diff --git a/automation/snippets/templates/rust/Cargo.toml b/automation/snippets/templates/rust/Cargo.toml index 7d93bd658..cdd358f9c 100644 --- a/automation/snippets/templates/rust/Cargo.toml +++ b/automation/snippets/templates/rust/Cargo.toml @@ -15,4 +15,5 @@ serde_json = "1.0.145" tempfile = "3" tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] } ureq = { version = "3", features = ["json"] } -uuid = { version = "1.18.1", features = ["v4"] } +uuid = { version = "1.18.1", features = ["v4", "v5"] } +sha2 = "0.11" diff --git a/qdrant-landing/content/documentation/headless/content/tutorials/operations.md b/qdrant-landing/content/documentation/headless/content/tutorials/operations.md index ab4967d0c..40907717e 100644 --- a/qdrant-landing/content/documentation/headless/content/tutorials/operations.md +++ b/qdrant-landing/content/documentation/headless/content/tutorials/operations.md @@ -6,6 +6,6 @@ | [Time-Based Sharding](/documentation/tutorials-operations/time-based-sharding/) | Efficiently manage time-series data with user-defined sharding. | Any | 1h | Intermediate | | [Large-Scale Search](/documentation/tutorials-operations/large-scale-search/) | Cost-efficient search for LAION-400M datasets. | Any | 48h | Advanced | | [Secure a Self-Hosted Instance](/documentation/tutorials-operations/secure-qdrant/) | Enable TLS, API keys, and JWT access control. | Any | 45m | Intermediate | -| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | Python | 25m | Beginner | +| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | Any | 25m | Beginner | | [Qdrant Cloud Prometheus Monitoring](/documentation/ops-monitoring/managed-cloud-prometheus/) | Observability with Prometheus and Grafana. | Prometheus | 30m | Intermediate | | [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | Prometheus | 30m | Intermediate | \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs new file mode 100644 index 000000000..69a233b8f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs @@ -0,0 +1,310 @@ +using System.Security.Cryptography; +using System.Text; +using System.Text.RegularExpressions; +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; +using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId); + +public class Snippet +{ + public static async Task Run() + { + // @block-start client-connection + var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL"); + var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY"); + + var client = new QdrantClient( + host: QDRANT_URL!, + https: true, + apiKey: QDRANT_API_KEY + ); + // @block-end client-connection + + // @hide-start + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and Normalize() live in the tutorial notebook + var CHUNKS = new List + { + ( + Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "prerequisites", + ChunkNum: 0, + Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + SectionUrl: "", ContentHash: "", PointId: "" + ), + ( + Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "step-3-enable-an-admin-api-key", + ChunkNum: 0, + Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + SectionUrl: "", ContentHash: "", PointId: "" + ), + }; + + string Normalize(string text) => Regex.Replace(text, @"\s+", " ").Trim(); + // @hide-end + + // @block-start create-collection + var MODEL = "sentence-transformers/all-MiniLM-L6-v2"; + var PIPELINE = "docs-prep-pipeline-v1"; + var COLLECTION = "docs-sync-tutorial"; + + await client.CreateCollectionAsync( + collectionName: COLLECTION, + vectorsConfig: new VectorParams + { + Size = 384, // all-MiniLM-L6-v2 output dimension + Distance = Distance.Cosine + }, + metadata: new() + { + ["embedding_model"] = MODEL, + ["pipeline_version"] = PIPELINE + } + ); + // @block-end create-collection + + // @block-start check-gate + async Task CheckGate() + { + // compare this pipeline's constants against what the collection records about itself + var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata; + var model = meta.GetValueOrDefault("embedding_model")?.StringValue; + var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue; + + if (model != MODEL || pipeline != PIPELINE) + throw new InvalidOperationException( + $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required"); + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + string ContentHash(string text) => + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant(); + + // Qdrant accepts any well-formed UUID as a point ID: + // a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID + string PointIdFor(string url, string anchor, int num) => + new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString(); + + // Derive both values (and the section address) for every raw chunk. + List PrepareChunksForSync(List chunks) + { + var prepared = new List(); + foreach (var c in chunks) + { + var text = Normalize(c.Text); + prepared.Add(c with + { + Text = text, + SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url, + ContentHash = ContentHash(text), + PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum), + }); + } + return prepared; + } + // @block-end identity-and-fingerprint + + // @block-start payload + Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new() + { + ["url"] = chunk.Url, + ["anchor"] = chunk.Anchor, + ["chunk_num"] = chunk.ChunkNum, + ["section_url"] = chunk.SectionUrl, + ["text"] = chunk.Text, + ["content_hash"] = chunk.ContentHash, + ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"), + }; + // @block-end payload + + // @block-start payload-indexes + foreach (var field in new[] { "content_hash", "url", "section_url" }) + await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword); + // @block-end payload-indexes + + // @block-start populate + await client.UpsertAsync( + collectionName: COLLECTION, + points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); + // @block-end populate + + // @block-start search + var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + + await client.QueryAsync( + collectionName: COLLECTION, + query: new Document { Text = QUERY, Model = MODEL }, + limit: 3, + payloadSelector: new[] { "section_url", "text" } + ); + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + var LATEST_CHUNKS = PrepareChunksForSync(CHUNKS); + // @hide-end + + // @block-start split-by-state + // Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. + async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)> + SplitByState(List latestChunks) + { + var incoming = latestChunks.ToDictionary(c => c.PointId); + + var stored = new Dictionary(); + var points = await client.RetrieveAsync( + COLLECTION, + ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(), + payloadSelector: new[] { "content_hash" }, + vectorSelector: false + ); + foreach (var p in points) + stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue; + + var unchanged = new List(); + var contentChanged = new List(); + var unknownIds = new List(); + foreach (var (pid, c) in incoming) + { + if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash) + unchanged.Add(c); + else if (stored.ContainsKey(pid)) + contentChanged.Add(c); + else + unknownIds.Add(c); + } + + return (incoming, unchanged, contentChanged, unknownIds); + } + + var splitState = await SplitByState(LATEST_CHUNKS); + // @block-end split-by-state + + // @block-start re-embed-changed + async Task ReEmbedChanged(List contentChanged) + { + if (contentChanged.Count == 0) + return; + await client.UpsertAsync( + collectionName: COLLECTION, + points: contentChanged.Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + // Reuse an existing embedding when the same text is already stored; embed only what is new. + async Task<(int reused, int added)> ReuseOrAdd(List unknownIds) + { + int reused = 0, added = 0; + + foreach (var c in unknownIds) + { + var sameText = new Filter + { + Must = { MatchKeyword("content_hash", c.ContentHash) } + }; + var hits = (await client.ScrollAsync( + COLLECTION, + filter: sameText, + limit: 1, + payloadSelector: new[] { "last_updated" }, + vectorsSelector: true + )).Result; + + PointStruct point; + if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(), + Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) }, + }; + reused++; + } + else // genuinely new content: embed and insert + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }; + added++; + } + + await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true); + } + + return (reused, added); + } + // @block-end reuse-or-add + + // @block-start delete-gone + // Remove every point the current crawl no longer contains. Returns how many. + async Task DeleteGone(Dictionary incomingIds) + { + if (incomingIds.Count == 0) + throw new ArgumentException("Refusing to delete from an empty source snapshot."); + + var stale = new Filter + { + MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) } + }; + + var toDelete = await client.CountAsync(COLLECTION, filter: stale); + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.DeleteAsync(COLLECTION, filter: stale, wait: true); + return toDelete; + } + // @block-end delete-gone + + // @block-start sync + async Task> Sync(List latestChunks) + { + await CheckGate(); // refuse to mix embedding models or pipeline versions + + var chunks = PrepareChunksForSync(latestChunks); + var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks); + + await ReEmbedChanged(contentChanged); + var (reused, added) = await ReuseOrAdd(unknownIds); + var deleted = await DeleteGone(incomingIds); + + return new Dictionary + { + ["unchanged"] = unchanged.Count, + ["re-embedded"] = contentChanged.Count, + ["reused_embedding"] = reused, + ["added"] = added, + ["deleted"] = (long)deleted, + }; + } + // @block-end sync + + // @block-start run-sync + var run = await Sync(LATEST_CHUNKS); + foreach (var (op, count) in run) + Console.WriteLine($"{op}: {count}"); + // @block-end run-sync + } + +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md new file mode 100644 index 000000000..703ed1765 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md @@ -0,0 +1,13 @@ +```csharp +async Task CheckGate() +{ + // compare this pipeline's constants against what the collection records about itself + var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata; + var model = meta.GetValueOrDefault("embedding_model")?.StringValue; + var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue; + + if (model != MODEL || pipeline != PIPELINE) + throw new InvalidOperationException( + $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required"); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md new file mode 100644 index 000000000..8a6cc5576 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md @@ -0,0 +1,11 @@ +```go +checkGate := func() { + // compare this pipeline's constants against what the collection records about itself + info, err := client.GetCollectionInfo(context.Background(), COLLECTION) + meta := info.GetConfig().GetMetadata() + + if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE { + panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta)) + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md new file mode 100644 index 000000000..a26a1300c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md @@ -0,0 +1,15 @@ +```java +static void checkGate() throws Exception { + // compare this pipeline's constants against what the collection records about itself + Map meta = + client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap(); + + Value model = meta.get("embedding_model"); + Value pipeline = meta.get("pipeline_version"); + if (model == null || !MODEL.equals(model.getStringValue()) + || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) { + throw new RuntimeException( + "collection was built by " + meta + ": full re-embed into a fresh collection required"); + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md new file mode 100644 index 000000000..bea57922a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md @@ -0,0 +1,8 @@ +```python +def check_gate(): + # compare this pipeline's constants against what the collection records about itself + meta = client.get_collection(COLLECTION).config.metadata or {} + + if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE: + raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required") +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md new file mode 100644 index 000000000..5e73ec86e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md @@ -0,0 +1,22 @@ +```rust +async fn check_gate(client: &Qdrant) -> anyhow::Result<()> { + // compare this pipeline's constants against what the collection records about itself + let meta = client + .collection_info(COLLECTION) + .await? + .result + .and_then(|info| info.config) + .map(|config| config.metadata) + .unwrap_or_default(); + + if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL) + || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str) + != Some(PIPELINE) + { + anyhow::bail!( + "collection was built by {meta:?}: full re-embed into a fresh collection required" + ); + } + Ok(()) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md new file mode 100644 index 000000000..9c2037820 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md @@ -0,0 +1,11 @@ +```typescript +async function checkGate() { + // compare this pipeline's constants against what the collection records about itself + const meta = ((await client.getCollection(COLLECTION)).config.metadata ?? + {}) as Record; + + if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) { + throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`); + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md new file mode 100644 index 000000000..9fef76a9a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md @@ -0,0 +1,10 @@ +```csharp +var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL"); +var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY"); + +var client = new QdrantClient( + host: QDRANT_URL!, + https: true, + apiKey: QDRANT_API_KEY +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md new file mode 100644 index 000000000..4f5465d1f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md @@ -0,0 +1,10 @@ +```go +QDRANT_URL := os.Getenv("QDRANT_URL") +QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY") + +client, err := qdrant.NewClient(&qdrant.Config{ + Host: QDRANT_URL, + APIKey: QDRANT_API_KEY, + UseTLS: true, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md new file mode 100644 index 000000000..89790dd9c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md @@ -0,0 +1,10 @@ +```java +static final String QDRANT_URL = System.getenv("QDRANT_URL"); +static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY"); + +static final QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true) + .withApiKey(QDRANT_API_KEY) + .build()); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md new file mode 100644 index 000000000..1ffb46824 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md @@ -0,0 +1,14 @@ +```python +import os + +from qdrant_client import QdrantClient, models + +QDRANT_URL = os.getenv("QDRANT_URL") +QDRANT_API_KEY = os.getenv("QDRANT_API_KEY") + +client = QdrantClient( + url=QDRANT_URL, + api_key=QDRANT_API_KEY, + cloud_inference=True +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md new file mode 100644 index 000000000..2ac1066f5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md @@ -0,0 +1,8 @@ +```rust +let qdrant_url = std::env::var("QDRANT_URL")?; +let qdrant_api_key = std::env::var("QDRANT_API_KEY")?; + +let client = Qdrant::from_url(&qdrant_url) + .api_key(qdrant_api_key) + .build()?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md new file mode 100644 index 000000000..23fe1f22a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md @@ -0,0 +1,11 @@ +```typescript +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +const QDRANT_URL = process.env.QDRANT_URL; +const QDRANT_API_KEY = process.env.QDRANT_API_KEY; + +const client = new QdrantClient({ + url: QDRANT_URL, + apiKey: QDRANT_API_KEY, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md new file mode 100644 index 000000000..cf9d7c8e3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md @@ -0,0 +1,19 @@ +```csharp +var MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +var PIPELINE = "docs-prep-pipeline-v1"; +var COLLECTION = "docs-sync-tutorial"; + +await client.CreateCollectionAsync( + collectionName: COLLECTION, + vectorsConfig: new VectorParams + { + Size = 384, // all-MiniLM-L6-v2 output dimension + Distance = Distance.Cosine + }, + metadata: new() + { + ["embedding_model"] = MODEL, + ["pipeline_version"] = PIPELINE + } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md new file mode 100644 index 000000000..daeb32756 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md @@ -0,0 +1,17 @@ +```go +MODEL := "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE := "docs-prep-pipeline-v1" +COLLECTION := "docs-sync-tutorial" + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: COLLECTION, + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 384, // all-MiniLM-L6-v2 output dimension + Distance: qdrant.Distance_Cosine, + }), + Metadata: qdrant.NewValueMap(map[string]any{ + "embedding_model": MODEL, + "pipeline_version": PIPELINE, + }), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md new file mode 100644 index 000000000..a41a81873 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md @@ -0,0 +1,24 @@ +```java +static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +static final String PIPELINE = "docs-prep-pipeline-v1"; +static final String COLLECTION = "docs-sync-tutorial"; + +static void createCollection() throws Exception { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(COLLECTION) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(384) // all-MiniLM-L6-v2 output dimension + .setDistance(Distance.Cosine) + .build()) + .build()) + .putAllMetadata( + Map.of( + "embedding_model", value(MODEL), + "pipeline_version", value(PIPELINE))) + .build()).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md new file mode 100644 index 000000000..5f350d2c6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md @@ -0,0 +1,14 @@ +```python +MODEL = "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE = "docs-prep-pipeline-v1" +COLLECTION = "docs-sync-tutorial" + +client.create_collection( + COLLECTION, + vectors_config=models.VectorParams( + size=384, # all-MiniLM-L6-v2 output dimension + distance=models.Distance.COSINE, + ), + metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE}, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md new file mode 100644 index 000000000..2a6853b45 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md @@ -0,0 +1,20 @@ +```rust +const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE: &str = "docs-prep-pipeline-v1"; +const COLLECTION: &str = "docs-sync-tutorial"; + +let mut metadata: HashMap = HashMap::new(); +metadata.insert("embedding_model".to_string(), json!(MODEL)); +metadata.insert("pipeline_version".to_string(), json!(PIPELINE)); + +client + .create_collection( + CreateCollectionBuilder::new(COLLECTION) + .vectors_config(VectorParamsBuilder::new( + 384, // all-MiniLM-L6-v2 output dimension + Distance::Cosine, + )) + .metadata(metadata), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md new file mode 100644 index 000000000..974e1777a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md @@ -0,0 +1,16 @@ +```typescript +const MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE = "docs-prep-pipeline-v1"; +const COLLECTION = "docs-sync-tutorial"; + +await client.createCollection(COLLECTION, { + vectors: { + size: 384, // all-MiniLM-L6-v2 output dimension + distance: "Cosine", + }, +}); + +await client.updateCollection(COLLECTION, { + metadata: { embedding_model: MODEL, pipeline_version: PIPELINE }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md new file mode 100644 index 000000000..860539841 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md @@ -0,0 +1,248 @@ +```csharp +using System.Security.Cryptography; +using System.Text; +using System.Text.RegularExpressions; +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; +using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId); + +var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL"); +var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY"); + +var client = new QdrantClient( + host: QDRANT_URL!, + https: true, + apiKey: QDRANT_API_KEY +); + +var MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +var PIPELINE = "docs-prep-pipeline-v1"; +var COLLECTION = "docs-sync-tutorial"; + +await client.CreateCollectionAsync( + collectionName: COLLECTION, + vectorsConfig: new VectorParams + { + Size = 384, // all-MiniLM-L6-v2 output dimension + Distance = Distance.Cosine + }, + metadata: new() + { + ["embedding_model"] = MODEL, + ["pipeline_version"] = PIPELINE + } +); + +async Task CheckGate() +{ + // compare this pipeline's constants against what the collection records about itself + var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata; + var model = meta.GetValueOrDefault("embedding_model")?.StringValue; + var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue; + + if (model != MODEL || pipeline != PIPELINE) + throw new InvalidOperationException( + $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required"); +} + +string ContentHash(string text) => + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant(); + +// Qdrant accepts any well-formed UUID as a point ID: +// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID +string PointIdFor(string url, string anchor, int num) => + new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString(); + +// Derive both values (and the section address) for every raw chunk. +List PrepareChunksForSync(List chunks) +{ + var prepared = new List(); + foreach (var c in chunks) + { + var text = Normalize(c.Text); + prepared.Add(c with + { + Text = text, + SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url, + ContentHash = ContentHash(text), + PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum), + }); + } + return prepared; +} + +Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new() +{ + ["url"] = chunk.Url, + ["anchor"] = chunk.Anchor, + ["chunk_num"] = chunk.ChunkNum, + ["section_url"] = chunk.SectionUrl, + ["text"] = chunk.Text, + ["content_hash"] = chunk.ContentHash, + ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"), +}; + +foreach (var field in new[] { "content_hash", "url", "section_url" }) + await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword); + +await client.UpsertAsync( + collectionName: COLLECTION, + points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true +); + +var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.QueryAsync( + collectionName: COLLECTION, + query: new Document { Text = QUERY, Model = MODEL }, + limit: 3, + payloadSelector: new[] { "section_url", "text" } +); + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)> + SplitByState(List latestChunks) +{ + var incoming = latestChunks.ToDictionary(c => c.PointId); + + var stored = new Dictionary(); + var points = await client.RetrieveAsync( + COLLECTION, + ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(), + payloadSelector: new[] { "content_hash" }, + vectorSelector: false + ); + foreach (var p in points) + stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue; + + var unchanged = new List(); + var contentChanged = new List(); + var unknownIds = new List(); + foreach (var (pid, c) in incoming) + { + if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash) + unchanged.Add(c); + else if (stored.ContainsKey(pid)) + contentChanged.Add(c); + else + unknownIds.Add(c); + } + + return (incoming, unchanged, contentChanged, unknownIds); +} + +var splitState = await SplitByState(LATEST_CHUNKS); + +async Task ReEmbedChanged(List contentChanged) +{ + if (contentChanged.Count == 0) + return; + await client.UpsertAsync( + collectionName: COLLECTION, + points: contentChanged.Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); +} + +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async Task<(int reused, int added)> ReuseOrAdd(List unknownIds) +{ + int reused = 0, added = 0; + + foreach (var c in unknownIds) + { + var sameText = new Filter + { + Must = { MatchKeyword("content_hash", c.ContentHash) } + }; + var hits = (await client.ScrollAsync( + COLLECTION, + filter: sameText, + limit: 1, + payloadSelector: new[] { "last_updated" }, + vectorsSelector: true + )).Result; + + PointStruct point; + if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(), + Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) }, + }; + reused++; + } + else // genuinely new content: embed and insert + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }; + added++; + } + + await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true); + } + + return (reused, added); +} + +// Remove every point the current crawl no longer contains. Returns how many. +async Task DeleteGone(Dictionary incomingIds) +{ + if (incomingIds.Count == 0) + throw new ArgumentException("Refusing to delete from an empty source snapshot."); + + var stale = new Filter + { + MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) } + }; + + var toDelete = await client.CountAsync(COLLECTION, filter: stale); + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.DeleteAsync(COLLECTION, filter: stale, wait: true); + return toDelete; +} + +async Task> Sync(List latestChunks) +{ + await CheckGate(); // refuse to mix embedding models or pipeline versions + + var chunks = PrepareChunksForSync(latestChunks); + var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks); + + await ReEmbedChanged(contentChanged); + var (reused, added) = await ReuseOrAdd(unknownIds); + var deleted = await DeleteGone(incomingIds); + + return new Dictionary + { + ["unchanged"] = unchanged.Count, + ["re-embedded"] = contentChanged.Count, + ["reused_embedding"] = reused, + ["added"] = added, + ["deleted"] = (long)deleted, + }; +} + +var run = await Sync(LATEST_CHUNKS); +foreach (var (op, count) in run) + Console.WriteLine($"{op}: {count}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md new file mode 100644 index 000000000..fd35a09f5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md @@ -0,0 +1,19 @@ +```csharp +// Remove every point the current crawl no longer contains. Returns how many. +async Task DeleteGone(Dictionary incomingIds) +{ + if (incomingIds.Count == 0) + throw new ArgumentException("Refusing to delete from an empty source snapshot."); + + var stale = new Filter + { + MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) } + }; + + var toDelete = await client.CountAsync(COLLECTION, filter: stale); + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.DeleteAsync(COLLECTION, filter: stale, wait: true); + return toDelete; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md new file mode 100644 index 000000000..9e915445f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md @@ -0,0 +1,29 @@ +```go +// remove every point the current crawl no longer contains, return how many +deleteGone := func(incomingIDs map[string]Chunk) int { + if len(incomingIDs) == 0 { + panic("Refusing to delete from an empty source snapshot.") + } + + ids := make([]*qdrant.PointId, 0, len(incomingIDs)) + for pid := range incomingIDs { + ids = append(ids, qdrant.NewID(pid)) + } + stale := &qdrant.Filter{ + MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)}, + } + + toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{ + CollectionName: COLLECTION, + Filter: stale, + }) + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.Delete(context.Background(), &qdrant.DeletePoints{ + CollectionName: COLLECTION, + Points: qdrant.NewPointsSelectorFilter(stale), + Wait: qdrant.PtrOf(true), + }) + return int(toDelete) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md new file mode 100644 index 000000000..724a5ef4a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md @@ -0,0 +1,21 @@ +```java +// Remove every point the current crawl no longer contains. Returns how many. +static long deleteGone(Map incomingIds) throws Exception { + if (incomingIds.isEmpty()) { + throw new IllegalArgumentException("Refusing to delete from an empty source snapshot."); + } + + Filter stale = Filter.newBuilder() + .addMustNot(hasId( + incomingIds.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()))) + .build(); + + long toDelete = client.countAsync(COLLECTION, stale, true).get(); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.deleteAsync(COLLECTION, stale).get(); + return toDelete; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md new file mode 100644 index 000000000..4a248f584 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md @@ -0,0 +1,14 @@ +```python +def delete_gone(incoming_ids): + """Remove every point the current crawl no longer contains. Returns how many.""" + if not incoming_ids: + raise ValueError("Refusing to delete from an empty source snapshot.") + + stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))]) + + to_delete = client.count(COLLECTION, count_filter=stale).count + + # potential check against a threshold to avoid accidental mass deletion could be added here + client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True) + return to_delete +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md new file mode 100644 index 000000000..d2d78da75 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md @@ -0,0 +1,28 @@ +```rust +/// Remove every point the current crawl no longer contains. Returns how many. +async fn delete_gone( + client: &Qdrant, + incoming_ids: &HashMap, +) -> anyhow::Result { + if incoming_ids.is_empty() { + anyhow::bail!("Refusing to delete from an empty source snapshot."); + } + + let stale = Filter::must_not([Condition::has_id( + incoming_ids.keys().map(|id| PointId::from(id.as_str())), + )]); + + let to_delete = client + .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone())) + .await? + .result + .map(|r| r.count) + .unwrap_or(0); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client + .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true)) + .await?; + Ok(to_delete) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md new file mode 100644 index 000000000..59a7dd039 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md @@ -0,0 +1,16 @@ +```typescript +// Remove every point the current crawl no longer contains. Returns how many. +async function deleteGone(incoming: Map) { + if (incoming.size === 0) { + throw new Error("Refusing to delete from an empty source snapshot."); + } + + const stale = { must_not: [{ has_id: [...incoming.keys()] }] }; + + const toDelete = (await client.count(COLLECTION, { filter: stale })).count; + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.delete(COLLECTION, { filter: stale, wait: true }); + return toDelete; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md new file mode 100644 index 000000000..9b310ee1e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md @@ -0,0 +1,276 @@ +```go +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "os" + "regexp" + "strings" + "time" + + "github.com/google/uuid" + "github.com/qdrant/go-client/qdrant" +) + +QDRANT_URL := os.Getenv("QDRANT_URL") +QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY") + +client, err := qdrant.NewClient(&qdrant.Config{ + Host: QDRANT_URL, + APIKey: QDRANT_API_KEY, + UseTLS: true, +}) + +MODEL := "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE := "docs-prep-pipeline-v1" +COLLECTION := "docs-sync-tutorial" + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: COLLECTION, + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 384, // all-MiniLM-L6-v2 output dimension + Distance: qdrant.Distance_Cosine, + }), + Metadata: qdrant.NewValueMap(map[string]any{ + "embedding_model": MODEL, + "pipeline_version": PIPELINE, + }), +}) + +checkGate := func() { + // compare this pipeline's constants against what the collection records about itself + info, err := client.GetCollectionInfo(context.Background(), COLLECTION) + meta := info.GetConfig().GetMetadata() + + if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE { + panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta)) + } +} + +contentHash := func(text string) string { + sum := sha256.Sum256([]byte(text)) + return hex.EncodeToString(sum[:]) +} + +pointID := func(url, anchor string, num int) string { + // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires, + // marking the input as a URL-like name + return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String() +} + +// derive both values (and the section address) for every raw chunk +prepareChunksForSync := func(chunks []Chunk) []Chunk { + out := make([]Chunk, 0, len(chunks)) + for _, c := range chunks { + c.Text = normalize(c.Text) + c.SectionURL = c.URL + if c.Anchor != "" { + c.SectionURL = c.URL + "#" + c.Anchor + } + c.ContentHash = contentHash(c.Text) + c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum) + out = append(out, c) + } + return out +} + +payload := func(c Chunk, lastUpdated string) map[string]any { + if lastUpdated == "" { + lastUpdated = time.Now().UTC().Format(time.RFC3339) + } + return map[string]any{ + "url": c.URL, + "anchor": c.Anchor, + "chunk_num": c.ChunkNum, + "section_url": c.SectionURL, + "text": c.Text, + "content_hash": c.ContentHash, + "last_updated": lastUpdated, + } +} + +for _, field := range []string{"content_hash", "url", "section_url"} { + client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: COLLECTION, + FieldName: field, + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + }) +} + +var points []*qdrant.PointStruct +for _, c := range prepareChunksForSync(CHUNKS) { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) +} +client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), +}) + +QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: COLLECTION, + Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}), + Limit: qdrant.PtrOf(uint64(3)), + WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"), +}) + +// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown +splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) { + incoming := make(map[string]Chunk, len(latestChunks)) + ids := make([]*qdrant.PointId, 0, len(latestChunks)) + for _, c := range latestChunks { + incoming[c.PointID] = c + ids = append(ids, qdrant.NewID(c.PointID)) + } + + retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{ + CollectionName: COLLECTION, + Ids: ids, + WithPayload: qdrant.NewWithPayloadInclude("content_hash"), + WithVectors: qdrant.NewWithVectors(false), + }) + stored := make(map[string]string, len(retrieved)) + for _, p := range retrieved { + stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue() + } + + var unchanged, contentChanged, unknownIDs []Chunk + for pid, c := range incoming { + storedHash, found := stored[pid] + switch { + case found && storedHash == c.ContentHash: + unchanged = append(unchanged, c) + case found: + contentChanged = append(contentChanged, c) + default: + unknownIDs = append(unknownIDs, c) + } + } + + return incoming, unchanged, contentChanged, unknownIDs +} + +incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS) + +reEmbedChanged := func(contentChanged []Chunk) { + if len(contentChanged) == 0 { + return + } + points := make([]*qdrant.PointStruct, 0, len(contentChanged)) + for _, c := range contentChanged { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) +} + +// reuse an existing embedding when the same text is already stored; embed only what is new +reuseOrAdd := func(unknownIDs []Chunk) (int, int) { + reused, added := 0, 0 + + for _, c := range unknownIDs { + sameText := &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("content_hash", c.ContentHash), + }, + } + hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: COLLECTION, + Filter: sameText, + Limit: qdrant.PtrOf(uint32(1)), + WithPayload: qdrant.NewWithPayloadInclude("last_updated"), + WithVectors: qdrant.NewWithVectors(true), + }) + + var point *qdrant.PointStruct + if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...), + Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())), + } + reused++ + } else { // genuinely new content: embed and insert + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + } + added++ + } + + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: []*qdrant.PointStruct{point}, + Wait: qdrant.PtrOf(true), + }) + } + + return reused, added +} + +// remove every point the current crawl no longer contains, return how many +deleteGone := func(incomingIDs map[string]Chunk) int { + if len(incomingIDs) == 0 { + panic("Refusing to delete from an empty source snapshot.") + } + + ids := make([]*qdrant.PointId, 0, len(incomingIDs)) + for pid := range incomingIDs { + ids = append(ids, qdrant.NewID(pid)) + } + stale := &qdrant.Filter{ + MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)}, + } + + toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{ + CollectionName: COLLECTION, + Filter: stale, + }) + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.Delete(context.Background(), &qdrant.DeletePoints{ + CollectionName: COLLECTION, + Points: qdrant.NewPointsSelectorFilter(stale), + Wait: qdrant.PtrOf(true), + }) + return int(toDelete) +} + +sync := func(latestChunks []Chunk) map[string]int { + checkGate() // refuse to mix embedding models or pipeline versions + + chunks := prepareChunksForSync(latestChunks) + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks) + + reEmbedChanged(contentChanged) + reused, added := reuseOrAdd(unknownIDs) + deleted := deleteGone(incomingIDs) + + return map[string]int{ + "unchanged": len(unchanged), + "re-embedded": len(contentChanged), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +} + +run := sync(LATEST_CHUNKS) +fmt.Println(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md new file mode 100644 index 000000000..9d7aced2a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md @@ -0,0 +1,27 @@ +```csharp +string ContentHash(string text) => + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant(); + +// Qdrant accepts any well-formed UUID as a point ID: +// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID +string PointIdFor(string url, string anchor, int num) => + new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString(); + +// Derive both values (and the section address) for every raw chunk. +List PrepareChunksForSync(List chunks) +{ + var prepared = new List(); + foreach (var c in chunks) + { + var text = Normalize(c.Text); + prepared.Add(c with + { + Text = text, + SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url, + ContentHash = ContentHash(text), + PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum), + }); + } + return prepared; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md new file mode 100644 index 000000000..4e89d7299 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md @@ -0,0 +1,28 @@ +```go +contentHash := func(text string) string { + sum := sha256.Sum256([]byte(text)) + return hex.EncodeToString(sum[:]) +} + +pointID := func(url, anchor string, num int) string { + // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires, + // marking the input as a URL-like name + return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String() +} + +// derive both values (and the section address) for every raw chunk +prepareChunksForSync := func(chunks []Chunk) []Chunk { + out := make([]Chunk, 0, len(chunks)) + for _, c := range chunks { + c.Text = normalize(c.Text) + c.SectionURL = c.URL + if c.Anchor != "" { + c.SectionURL = c.URL + "#" + c.Anchor + } + c.ContentHash = contentHash(c.Text) + c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum) + out = append(out, c) + } + return out +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md new file mode 100644 index 000000000..b98115908 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md @@ -0,0 +1,27 @@ +```java +static String contentHash(String text) throws Exception { + byte[] digest = MessageDigest.getInstance("SHA-256") + .digest(text.getBytes(StandardCharsets.UTF_8)); + return String.format("%064x", new BigInteger(1, digest)); +} + +static String pointId(String url, String anchor, int num) { + // name-based UUID (version 3); the same address always yields the same ID + return UUID.nameUUIDFromBytes( + (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString(); +} + +// Derive both values (and the section address) for every raw chunk. +static List prepareChunksForSync(List chunks) throws Exception { + List out = new ArrayList<>(); + for (Chunk c : chunks) { + String text = normalize(c.text); + Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text); + prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url; + prepared.contentHash = contentHash(text); + prepared.pointId = pointId(c.url, c.anchor, c.chunkNum); + out.add(prepared); + } + return out; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md new file mode 100644 index 000000000..13c467afb --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md @@ -0,0 +1,26 @@ +```python +import hashlib +import uuid +from datetime import datetime, timezone + +def content_hash(text): + return hashlib.sha256(text.encode()).hexdigest() + +def point_id(url, anchor, num): + # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}")) + +def prepare_chunks_for_sync(chunks): + """Derive both values (and the section address) for every raw chunk.""" + out = [] + for c in chunks: + text = normalize(c["text"]) + out.append({ + **c, + "text": text, + "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"], + "content_hash": content_hash(text), + "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]), + }) + return out +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md new file mode 100644 index 000000000..f14b3d7e1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md @@ -0,0 +1,38 @@ +```rust +fn content_hash(text: &str) -> String { + Sha256::digest(text.as_bytes()) + .iter() + .map(|byte| format!("{byte:02x}")) + .collect() +} + +fn point_id(url: &str, anchor: &str, num: u32) -> String { + // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + uuid::Uuid::new_v5( + &uuid::Uuid::NAMESPACE_URL, + format!("{url}#{anchor}::{num}").as_bytes(), + ) + .to_string() +} + +/// Derive both values (and the section address) for every raw chunk. +fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec { + chunks + .iter() + .map(|c| { + let text = normalize(&c.text); + Chunk { + text: text.clone(), + section_url: if c.anchor.is_empty() { + c.url.clone() + } else { + format!("{}#{}", c.url, c.anchor) + }, + content_hash: content_hash(&text), + point_id: point_id(&c.url, &c.anchor, c.chunk_num), + ..c.clone() + } + }) + .collect() +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md new file mode 100644 index 000000000..ae3486389 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md @@ -0,0 +1,32 @@ +```typescript +import { createHash } from "node:crypto"; + +type RawChunk = { url: string; anchor: string; chunk_num: number; text: string }; +type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string }; + +function contentHash(text: string): string { + return createHash("sha256").update(text).digest("hex"); +} + +// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name +function pointId(url: string, anchor: string, num: number): string { + // Qdrant accepts any well-formed UUID as a point ID: + // hash the address, format the digest as a UUID, and the same address always yields the same ID + const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex"); + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`; +} + +// Derive both values (and the section address) for every raw chunk. +function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] { + return chunks.map((c) => { + const text = normalize(c.text); + return { + ...c, + text, + section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url, + content_hash: contentHash(text), + point_id: pointId(c.url, c.anchor, c.chunk_num), + }; + }); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md new file mode 100644 index 000000000..e3bf0bcc6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md @@ -0,0 +1,328 @@ +```java +import static io.qdrant.client.ConditionFactory.hasId; +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.PointIdFactory.id; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorFactory.vector; +import static io.qdrant.client.VectorsFactory.vectors; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.VectorOutputHelper; +import io.qdrant.client.WithPayloadSelectorFactory; +import io.qdrant.client.WithVectorsSelectorFactory; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.JsonWithInt.Value; +import io.qdrant.client.grpc.Points.Document; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ScrollPoints; +import java.math.BigInteger; +import java.nio.charset.StandardCharsets; +import java.security.MessageDigest; +import java.time.OffsetDateTime; +import java.time.ZoneOffset; +import java.time.temporal.ChronoUnit; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import java.util.UUID; +import java.util.stream.Collectors; + +static final String QDRANT_URL = System.getenv("QDRANT_URL"); +static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY"); + +static final QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true) + .withApiKey(QDRANT_API_KEY) + .build()); + +static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +static final String PIPELINE = "docs-prep-pipeline-v1"; +static final String COLLECTION = "docs-sync-tutorial"; + +static void createCollection() throws Exception { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(COLLECTION) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(384) // all-MiniLM-L6-v2 output dimension + .setDistance(Distance.Cosine) + .build()) + .build()) + .putAllMetadata( + Map.of( + "embedding_model", value(MODEL), + "pipeline_version", value(PIPELINE))) + .build()).get(); +} + +static void checkGate() throws Exception { + // compare this pipeline's constants against what the collection records about itself + Map meta = + client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap(); + + Value model = meta.get("embedding_model"); + Value pipeline = meta.get("pipeline_version"); + if (model == null || !MODEL.equals(model.getStringValue()) + || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) { + throw new RuntimeException( + "collection was built by " + meta + ": full re-embed into a fresh collection required"); + } +} + +static String contentHash(String text) throws Exception { + byte[] digest = MessageDigest.getInstance("SHA-256") + .digest(text.getBytes(StandardCharsets.UTF_8)); + return String.format("%064x", new BigInteger(1, digest)); +} + +static String pointId(String url, String anchor, int num) { + // name-based UUID (version 3); the same address always yields the same ID + return UUID.nameUUIDFromBytes( + (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString(); +} + +// Derive both values (and the section address) for every raw chunk. +static List prepareChunksForSync(List chunks) throws Exception { + List out = new ArrayList<>(); + for (Chunk c : chunks) { + String text = normalize(c.text); + Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text); + prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url; + prepared.contentHash = contentHash(text); + prepared.pointId = pointId(c.url, c.anchor, c.chunkNum); + out.add(prepared); + } + return out; +} + +static Map payload(Chunk chunk, String lastUpdated) { + Map p = new HashMap<>(); + p.put("url", value(chunk.url)); + p.put("anchor", value(chunk.anchor)); + p.put("chunk_num", value(chunk.chunkNum)); + p.put("section_url", value(chunk.sectionUrl)); + p.put("text", value(chunk.text)); + p.put("content_hash", value(chunk.contentHash)); + p.put("last_updated", value(lastUpdated != null + ? lastUpdated + : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString())); + return p; +} + +static void createPayloadIndexes() throws Exception { + for (String field : List.of("content_hash", "url", "section_url")) { + client.createPayloadIndexAsync( + COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get(); + } +} + +static void populate() throws Exception { + List points = new ArrayList<>(); + for (Chunk c : prepareChunksForSync(CHUNKS)) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} + +static final String QUERY = + "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +static void search() throws Exception { + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(COLLECTION) + .setQuery( + nearest( + Document.newBuilder() + .setText(QUERY) + .setModel(MODEL) + .build())) + .setLimit(3) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text"))) + .build()).get(); +} + +static class SyncState { + Map incoming = new LinkedHashMap<>(); + List unchanged = new ArrayList<>(); + List contentChanged = new ArrayList<>(); + List unknownIds = new ArrayList<>(); +} + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +static SyncState splitByState(List latestChunks) throws Exception { + SyncState state = new SyncState(); + for (Chunk c : latestChunks) { + state.incoming.put(c.pointId, c); + } + + Map stored = new HashMap<>(); + var points = client.retrieveAsync( + COLLECTION, + state.incoming.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()), + WithPayloadSelectorFactory.include(List.of("content_hash")), + WithVectorsSelectorFactory.enable(false), + null).get(); + for (var p : points) { + stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue()); + } + + for (Map.Entry e : state.incoming.entrySet()) { + String pid = e.getKey(); + Chunk c = e.getValue(); + if (c.contentHash.equals(stored.get(pid))) { + state.unchanged.add(c); + } else if (stored.containsKey(pid)) { + state.contentChanged.add(c); + } else { + state.unknownIds.add(c); + } + } + + return state; +} + +static void reEmbedChanged(List contentChanged) throws Exception { + if (contentChanged.isEmpty()) { + return; + } + List points = new ArrayList<>(); + for (Chunk c : contentChanged) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} + +// Reuse an existing embedding when the same text is already stored; embed only what is new. +static int[] reuseOrAdd(List unknownIds) throws Exception { + int reused = 0; + int added = 0; + + for (Chunk c : unknownIds) { + Filter sameText = Filter.newBuilder() + .addMust(matchKeyword("content_hash", c.contentHash)) + .build(); + + var hits = client.scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName(COLLECTION) + .setFilter(sameText) + .setLimit(1) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated"))) + .setWithVectors(WithVectorsSelectorFactory.enable(true)) + .build()).get().getResultList(); + + PointStruct point; + if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors(vectors(vector( + VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector()) + .getDataList()))) + .putAllPayload( + payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue())) + .build(); + reused++; + } else { // genuinely new content: embed and insert + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build(); + added++; + } + + client.upsertAsync(COLLECTION, List.of(point)).get(); + } + + return new int[] {reused, added}; +} + +// Remove every point the current crawl no longer contains. Returns how many. +static long deleteGone(Map incomingIds) throws Exception { + if (incomingIds.isEmpty()) { + throw new IllegalArgumentException("Refusing to delete from an empty source snapshot."); + } + + Filter stale = Filter.newBuilder() + .addMustNot(hasId( + incomingIds.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()))) + .build(); + + long toDelete = client.countAsync(COLLECTION, stale, true).get(); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.deleteAsync(COLLECTION, stale).get(); + return toDelete; +} + +static Map sync(List latestChunks) throws Exception { + checkGate(); // refuse to mix embedding models or pipeline versions + + List chunks = prepareChunksForSync(latestChunks); + SyncState state = splitByState(chunks); + + reEmbedChanged(state.contentChanged); + int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added} + long deleted = deleteGone(state.incoming); + + return Map.of( + "unchanged", (long) state.unchanged.size(), + "re-embedded", (long) state.contentChanged.size(), + "reused_embedding", (long) reusedAdded[0], + "added", (long) reusedAdded[1], + "deleted", deleted); +} + +static void runSync() throws Exception { + Map run = sync(LATEST_CHUNKS); + System.out.println(run); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md new file mode 100644 index 000000000..ecf3434c6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md @@ -0,0 +1,4 @@ +```csharp +foreach (var field in new[] { "content_hash", "url", "section_url" }) + await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md new file mode 100644 index 000000000..f51c00a2e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md @@ -0,0 +1,9 @@ +```go +for _, field := range []string{"content_hash", "url", "section_url"} { + client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: COLLECTION, + FieldName: field, + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + }) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md new file mode 100644 index 000000000..9765e8e4c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md @@ -0,0 +1,8 @@ +```java +static void createPayloadIndexes() throws Exception { + for (String field : List.of("content_hash", "url", "section_url")) { + client.createPayloadIndexAsync( + COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get(); + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md new file mode 100644 index 000000000..9a0c2b8e4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md @@ -0,0 +1,4 @@ +```python +for field in ("content_hash", "url", "section_url"): + client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md new file mode 100644 index 000000000..388e1a49c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md @@ -0,0 +1,11 @@ +```rust +for field in ["content_hash", "url", "section_url"] { + client + .create_field_index(CreateFieldIndexCollectionBuilder::new( + COLLECTION, + field, + FieldType::Keyword, + )) + .await?; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md new file mode 100644 index 000000000..c2ac75c12 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md @@ -0,0 +1,8 @@ +```typescript +for (const field of ["content_hash", "url", "section_url"]) { + await client.createPayloadIndex(COLLECTION, { + field_name: field, + field_schema: "keyword", + }); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md new file mode 100644 index 000000000..d1cffd33c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md @@ -0,0 +1,12 @@ +```csharp +Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new() +{ + ["url"] = chunk.Url, + ["anchor"] = chunk.Anchor, + ["chunk_num"] = chunk.ChunkNum, + ["section_url"] = chunk.SectionUrl, + ["text"] = chunk.Text, + ["content_hash"] = chunk.ContentHash, + ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"), +}; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md new file mode 100644 index 000000000..253bb1df0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md @@ -0,0 +1,16 @@ +```go +payload := func(c Chunk, lastUpdated string) map[string]any { + if lastUpdated == "" { + lastUpdated = time.Now().UTC().Format(time.RFC3339) + } + return map[string]any{ + "url": c.URL, + "anchor": c.Anchor, + "chunk_num": c.ChunkNum, + "section_url": c.SectionURL, + "text": c.Text, + "content_hash": c.ContentHash, + "last_updated": lastUpdated, + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md new file mode 100644 index 000000000..d7e9ff1d2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md @@ -0,0 +1,15 @@ +```java +static Map payload(Chunk chunk, String lastUpdated) { + Map p = new HashMap<>(); + p.put("url", value(chunk.url)); + p.put("anchor", value(chunk.anchor)); + p.put("chunk_num", value(chunk.chunkNum)); + p.put("section_url", value(chunk.sectionUrl)); + p.put("text", value(chunk.text)); + p.put("content_hash", value(chunk.contentHash)); + p.put("last_updated", value(lastUpdated != null + ? lastUpdated + : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString())); + return p; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md new file mode 100644 index 000000000..3e8896b38 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md @@ -0,0 +1,12 @@ +```python +def payload(chunk, last_updated=None): + return { + "url": chunk["url"], + "anchor": chunk["anchor"], + "chunk_num": chunk["chunk_num"], + "section_url": chunk["section_url"], + "text": chunk["text"], + "content_hash": chunk["content_hash"], + "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"), + } +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md new file mode 100644 index 000000000..840abf040 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md @@ -0,0 +1,16 @@ +```rust +fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result { + let last_updated = last_updated.unwrap_or_else(|| { + chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false) + }); + Ok(Payload::try_from(serde_json::json!({ + "url": chunk.url, + "anchor": chunk.anchor, + "chunk_num": chunk.chunk_num, + "section_url": chunk.section_url, + "text": chunk.text, + "content_hash": chunk.content_hash, + "last_updated": last_updated, + }))?) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md new file mode 100644 index 000000000..f35654b03 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md @@ -0,0 +1,13 @@ +```typescript +function payload(chunk: SyncChunk, lastUpdated?: string) { + return { + url: chunk.url, + anchor: chunk.anchor, + chunk_num: chunk.chunk_num, + section_url: chunk.section_url, + text: chunk.text, + content_hash: chunk.content_hash, + last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"), + }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md new file mode 100644 index 000000000..15d9a0904 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md @@ -0,0 +1,12 @@ +```csharp +await client.UpsertAsync( + collectionName: COLLECTION, + points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md new file mode 100644 index 000000000..c6599239a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md @@ -0,0 +1,16 @@ +```go +var points []*qdrant.PointStruct +for _, c := range prepareChunksForSync(CHUNKS) { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) +} +client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md new file mode 100644 index 000000000..bddc2017f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md @@ -0,0 +1,21 @@ +```java +static void populate() throws Exception { + List points = new ArrayList<>(); + for (Chunk c : prepareChunksForSync(CHUNKS)) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md new file mode 100644 index 000000000..66ee0cfbc --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md @@ -0,0 +1,10 @@ +```python +client.upsert(COLLECTION, points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in prepare_chunks_for_sync(CHUNKS) +], wait=True) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md new file mode 100644 index 000000000..2bdd39c35 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md @@ -0,0 +1,16 @@ +```rust +let points: Vec = prepare_chunks_for_sync(&chunks) + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + +client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md new file mode 100644 index 000000000..b27ed319f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md @@ -0,0 +1,10 @@ +```typescript +await client.upsert(COLLECTION, { + points: prepareChunksForSync(CHUNKS).map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md new file mode 100644 index 000000000..e75201b84 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md @@ -0,0 +1,203 @@ +```python +import os + +from qdrant_client import QdrantClient, models + +QDRANT_URL = os.getenv("QDRANT_URL") +QDRANT_API_KEY = os.getenv("QDRANT_API_KEY") + +client = QdrantClient( + url=QDRANT_URL, + api_key=QDRANT_API_KEY, + cloud_inference=True +) + +MODEL = "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE = "docs-prep-pipeline-v1" +COLLECTION = "docs-sync-tutorial" + +client.create_collection( + COLLECTION, + vectors_config=models.VectorParams( + size=384, # all-MiniLM-L6-v2 output dimension + distance=models.Distance.COSINE, + ), + metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE}, +) + +def check_gate(): + # compare this pipeline's constants against what the collection records about itself + meta = client.get_collection(COLLECTION).config.metadata or {} + + if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE: + raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required") + +import hashlib +import uuid +from datetime import datetime, timezone + +def content_hash(text): + return hashlib.sha256(text.encode()).hexdigest() + +def point_id(url, anchor, num): + # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}")) + +def prepare_chunks_for_sync(chunks): + """Derive both values (and the section address) for every raw chunk.""" + out = [] + for c in chunks: + text = normalize(c["text"]) + out.append({ + **c, + "text": text, + "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"], + "content_hash": content_hash(text), + "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]), + }) + return out + +def payload(chunk, last_updated=None): + return { + "url": chunk["url"], + "anchor": chunk["anchor"], + "chunk_num": chunk["chunk_num"], + "section_url": chunk["section_url"], + "text": chunk["text"], + "content_hash": chunk["content_hash"], + "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"), + } + +for field in ("content_hash", "url", "section_url"): + client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD) + +client.upsert(COLLECTION, points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in prepare_chunks_for_sync(CHUNKS) +], wait=True) + +QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.query_points( + COLLECTION, + query=models.Document(text=QUERY, model=MODEL), + limit=3, + with_payload=["section_url", "text"], +) + +def split_by_state(latest_chunks): + """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.""" + incoming = {c["point_id"]: c for c in latest_chunks} + + stored = {} + points = client.retrieve( + COLLECTION, + ids=list(incoming), + with_payload=["content_hash"], + with_vectors=False, + ) + for p in points: + stored[str(p.id)] = p.payload["content_hash"] + + unchanged, content_changed, unknown_ids = [], [], [] + for pid, c in incoming.items(): + if stored.get(pid) == c["content_hash"]: + unchanged.append(c) + elif pid in stored: + content_changed.append(c) + else: + unknown_ids.append(c) + + return incoming, unchanged, content_changed, unknown_ids + +incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS) + +def re_embed_changed(content_changed): + if not content_changed: + return + client.upsert(COLLECTION, + points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in content_changed], + wait=True) + +def reuse_or_add(unknown_ids): + """Reuse an existing embedding when the same text is already stored; embed only what is new.""" + reused, added = 0, 0 + + for c in unknown_ids: + same_text = models.Filter(must=[ + models.FieldCondition( + key="content_hash", + match=models.MatchValue(value=c["content_hash"]), + ) + ]) + hits, _ = client.scroll( + COLLECTION, + scroll_filter=same_text, + limit=1, + with_payload=["last_updated"], + with_vectors=True, + ) + + if hits: # same text, new address: copy the vector, keep its last_updated + point = models.PointStruct( + id=c["point_id"], + vector=hits[0].vector, + payload=payload(c, hits[0].payload["last_updated"]), + ) + reused += 1 + else: # genuinely new content: embed and insert + point = models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + added += 1 + + client.upsert(COLLECTION, points=[point], wait=True) + + return reused, added + +def delete_gone(incoming_ids): + """Remove every point the current crawl no longer contains. Returns how many.""" + if not incoming_ids: + raise ValueError("Refusing to delete from an empty source snapshot.") + + stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))]) + + to_delete = client.count(COLLECTION, count_filter=stale).count + + # potential check against a threshold to avoid accidental mass deletion could be added here + client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True) + return to_delete + +def sync(latest_chunks): + check_gate() # refuse to mix embedding models or pipeline versions + + chunks = prepare_chunks_for_sync(latest_chunks) + incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks) + + re_embed_changed(content_changed) + reused, added = reuse_or_add(unknown_ids) + deleted = delete_gone(incoming_ids) + + return { + "unchanged": len(unchanged), + "re-embedded": len(content_changed), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } + +run = sync(LATEST_CHUNKS) +print(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md new file mode 100644 index 000000000..e7de2da10 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md @@ -0,0 +1,17 @@ +```csharp +async Task ReEmbedChanged(List contentChanged) +{ + if (contentChanged.Count == 0) + return; + await client.UpsertAsync( + collectionName: COLLECTION, + points: contentChanged.Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md new file mode 100644 index 000000000..a6a6d1cfd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md @@ -0,0 +1,20 @@ +```go +reEmbedChanged := func(contentChanged []Chunk) { + if len(contentChanged) == 0 { + return + } + points := make([]*qdrant.PointStruct, 0, len(contentChanged)) + for _, c := range contentChanged { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md new file mode 100644 index 000000000..a8b89be18 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md @@ -0,0 +1,23 @@ +```java +static void reEmbedChanged(List contentChanged) throws Exception { + if (contentChanged.isEmpty()) { + return; + } + List points = new ArrayList<>(); + for (Chunk c : contentChanged) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md new file mode 100644 index 000000000..a7bd926f4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md @@ -0,0 +1,14 @@ +```python +def re_embed_changed(content_changed): + if not content_changed: + return + client.upsert(COLLECTION, + points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in content_changed], + wait=True) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md new file mode 100644 index 000000000..c67faf875 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md @@ -0,0 +1,22 @@ +```rust +async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> { + if content_changed.is_empty() { + return Ok(()); + } + let points: Vec = content_changed + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + Ok(()) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md new file mode 100644 index 000000000..5f5b41e5c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md @@ -0,0 +1,15 @@ +```typescript +async function reEmbedChanged(contentChanged: SyncChunk[]) { + if (contentChanged.length === 0) { + return; + } + await client.upsert(COLLECTION, { + points: contentChanged.map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, + }); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md new file mode 100644 index 000000000..0be445308 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md @@ -0,0 +1,48 @@ +```csharp +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async Task<(int reused, int added)> ReuseOrAdd(List unknownIds) +{ + int reused = 0, added = 0; + + foreach (var c in unknownIds) + { + var sameText = new Filter + { + Must = { MatchKeyword("content_hash", c.ContentHash) } + }; + var hits = (await client.ScrollAsync( + COLLECTION, + filter: sameText, + limit: 1, + payloadSelector: new[] { "last_updated" }, + vectorsSelector: true + )).Result; + + PointStruct point; + if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(), + Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) }, + }; + reused++; + } + else // genuinely new content: embed and insert + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }; + added++; + } + + await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true); + } + + return (reused, added); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md new file mode 100644 index 000000000..524e8595c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md @@ -0,0 +1,46 @@ +```go +// reuse an existing embedding when the same text is already stored; embed only what is new +reuseOrAdd := func(unknownIDs []Chunk) (int, int) { + reused, added := 0, 0 + + for _, c := range unknownIDs { + sameText := &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("content_hash", c.ContentHash), + }, + } + hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: COLLECTION, + Filter: sameText, + Limit: qdrant.PtrOf(uint32(1)), + WithPayload: qdrant.NewWithPayloadInclude("last_updated"), + WithVectors: qdrant.NewWithVectors(true), + }) + + var point *qdrant.PointStruct + if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...), + Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())), + } + reused++ + } else { // genuinely new content: embed and insert + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + } + added++ + } + + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: []*qdrant.PointStruct{point}, + Wait: qdrant.PtrOf(true), + }) + } + + return reused, added +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md new file mode 100644 index 000000000..263901390 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md @@ -0,0 +1,52 @@ +```java +// Reuse an existing embedding when the same text is already stored; embed only what is new. +static int[] reuseOrAdd(List unknownIds) throws Exception { + int reused = 0; + int added = 0; + + for (Chunk c : unknownIds) { + Filter sameText = Filter.newBuilder() + .addMust(matchKeyword("content_hash", c.contentHash)) + .build(); + + var hits = client.scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName(COLLECTION) + .setFilter(sameText) + .setLimit(1) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated"))) + .setWithVectors(WithVectorsSelectorFactory.enable(true)) + .build()).get().getResultList(); + + PointStruct point; + if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors(vectors(vector( + VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector()) + .getDataList()))) + .putAllPayload( + payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue())) + .build(); + reused++; + } else { // genuinely new content: embed and insert + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build(); + added++; + } + + client.upsertAsync(COLLECTION, List.of(point)).get(); + } + + return new int[] {reused, added}; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md new file mode 100644 index 000000000..18edc2b34 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md @@ -0,0 +1,39 @@ +```python +def reuse_or_add(unknown_ids): + """Reuse an existing embedding when the same text is already stored; embed only what is new.""" + reused, added = 0, 0 + + for c in unknown_ids: + same_text = models.Filter(must=[ + models.FieldCondition( + key="content_hash", + match=models.MatchValue(value=c["content_hash"]), + ) + ]) + hits, _ = client.scroll( + COLLECTION, + scroll_filter=same_text, + limit=1, + with_payload=["last_updated"], + with_vectors=True, + ) + + if hits: # same text, new address: copy the vector, keep its last_updated + point = models.PointStruct( + id=c["point_id"], + vector=hits[0].vector, + payload=payload(c, hits[0].payload["last_updated"]), + ) + reused += 1 + else: # genuinely new content: embed and insert + point = models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + added += 1 + + client.upsert(COLLECTION, points=[point], wait=True) + + return reused, added +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md new file mode 100644 index 000000000..78bffc0e5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md @@ -0,0 +1,51 @@ +```rust +/// Reuse an existing embedding when the same text is already stored; embed only what is new. +async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> { + let (mut reused, mut added) = (0, 0); + + for c in unknown_ids { + let same_text = + Filter::must([Condition::matches("content_hash", c.content_hash.clone())]); + let hits = client + .scroll( + ScrollPointsBuilder::new(COLLECTION) + .filter(same_text) + .limit(1) + .with_payload(PayloadIncludeSelector::new(vec![ + "last_updated".to_string() + ])) + .with_vectors(true), + ) + .await? + .result; + + let point = if let Some(hit) = hits.into_iter().next() { + // same text, new address: copy the vector, keep its last_updated + let last_updated = hit.get("last_updated").as_str().cloned(); + let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) { + Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector { + Some(vector_output::Vector::Dense(dense)) => dense.data, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }; + reused += 1; + PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?) + } else { + // genuinely new content: embed and insert + added += 1; + PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + ) + }; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true)) + .await?; + } + + Ok((reused, added)) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md new file mode 100644 index 000000000..b7ac83ae2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md @@ -0,0 +1,45 @@ +```typescript +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async function reuseOrAdd(unknownIds: SyncChunk[]) { + let reused = 0; + let added = 0; + + for (const c of unknownIds) { + const sameText = { + must: [ + { + key: "content_hash", + match: { value: c.content_hash }, + }, + ], + }; + const hits = (await client.scroll(COLLECTION, { + filter: sameText, + limit: 1, + with_payload: ["last_updated"], + with_vector: true, + })).points; + + let point: Schemas["PointStruct"]; + if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated + point = { + id: c.point_id, + vector: hits[0].vector as number[], + payload: payload(c, hits[0].payload?.last_updated as string), + }; + reused += 1; + } else { // genuinely new content: embed and insert + point = { + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + }; + added += 1; + } + + await client.upsert(COLLECTION, { points: [point], wait: true }); + } + + return { reused, added }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md new file mode 100644 index 000000000..1d059fa63 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md @@ -0,0 +1,5 @@ +```csharp +var run = await Sync(LATEST_CHUNKS); +foreach (var (op, count) in run) + Console.WriteLine($"{op}: {count}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md new file mode 100644 index 000000000..e57dc2911 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md @@ -0,0 +1,4 @@ +```go +run := sync(LATEST_CHUNKS) +fmt.Println(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md new file mode 100644 index 000000000..f249c6a3e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md @@ -0,0 +1,6 @@ +```java +static void runSync() throws Exception { + Map run = sync(LATEST_CHUNKS); + System.out.println(run); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md new file mode 100644 index 000000000..e4e35c0fa --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md @@ -0,0 +1,4 @@ +```python +run = sync(LATEST_CHUNKS) +print(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md new file mode 100644 index 000000000..835e689d9 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md @@ -0,0 +1,4 @@ +```rust +let run = sync(&client, &latest_chunks).await?; +println!("{run:?}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md new file mode 100644 index 000000000..9fc1e07a1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md @@ -0,0 +1,4 @@ +```typescript +const run = await sync(LATEST_CHUNKS); +console.log(run); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md new file mode 100644 index 000000000..e5dcc9f69 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md @@ -0,0 +1,322 @@ +```rust +use serde_json::{json, Value}; +use std::collections::HashMap; + +use qdrant_client::qdrant::{ + point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder, + CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance, + Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct, + Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder, +}; +use qdrant_client::{Payload, Qdrant}; +use sha2::{Digest, Sha256}; + +let qdrant_url = std::env::var("QDRANT_URL")?; +let qdrant_api_key = std::env::var("QDRANT_API_KEY")?; + +let client = Qdrant::from_url(&qdrant_url) + .api_key(qdrant_api_key) + .build()?; + +const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE: &str = "docs-prep-pipeline-v1"; +const COLLECTION: &str = "docs-sync-tutorial"; + +let mut metadata: HashMap = HashMap::new(); +metadata.insert("embedding_model".to_string(), json!(MODEL)); +metadata.insert("pipeline_version".to_string(), json!(PIPELINE)); + +client + .create_collection( + CreateCollectionBuilder::new(COLLECTION) + .vectors_config(VectorParamsBuilder::new( + 384, // all-MiniLM-L6-v2 output dimension + Distance::Cosine, + )) + .metadata(metadata), + ) + .await?; + +async fn check_gate(client: &Qdrant) -> anyhow::Result<()> { + // compare this pipeline's constants against what the collection records about itself + let meta = client + .collection_info(COLLECTION) + .await? + .result + .and_then(|info| info.config) + .map(|config| config.metadata) + .unwrap_or_default(); + + if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL) + || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str) + != Some(PIPELINE) + { + anyhow::bail!( + "collection was built by {meta:?}: full re-embed into a fresh collection required" + ); + } + Ok(()) +} + +fn content_hash(text: &str) -> String { + Sha256::digest(text.as_bytes()) + .iter() + .map(|byte| format!("{byte:02x}")) + .collect() +} + +fn point_id(url: &str, anchor: &str, num: u32) -> String { + // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + uuid::Uuid::new_v5( + &uuid::Uuid::NAMESPACE_URL, + format!("{url}#{anchor}::{num}").as_bytes(), + ) + .to_string() +} + +/// Derive both values (and the section address) for every raw chunk. +fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec { + chunks + .iter() + .map(|c| { + let text = normalize(&c.text); + Chunk { + text: text.clone(), + section_url: if c.anchor.is_empty() { + c.url.clone() + } else { + format!("{}#{}", c.url, c.anchor) + }, + content_hash: content_hash(&text), + point_id: point_id(&c.url, &c.anchor, c.chunk_num), + ..c.clone() + } + }) + .collect() +} + +fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result { + let last_updated = last_updated.unwrap_or_else(|| { + chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false) + }); + Ok(Payload::try_from(serde_json::json!({ + "url": chunk.url, + "anchor": chunk.anchor, + "chunk_num": chunk.chunk_num, + "section_url": chunk.section_url, + "text": chunk.text, + "content_hash": chunk.content_hash, + "last_updated": last_updated, + }))?) +} + +for field in ["content_hash", "url", "section_url"] { + client + .create_field_index(CreateFieldIndexCollectionBuilder::new( + COLLECTION, + field, + FieldType::Keyword, + )) + .await?; +} + +let points: Vec = prepare_chunks_for_sync(&chunks) + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + +client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + +const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +client + .query( + QueryPointsBuilder::new(COLLECTION) + .query(Query::new_nearest(Document::new(QUERY, MODEL))) + .limit(3) + .with_payload(PayloadIncludeSelector::new(vec![ + "section_url".to_string(), + "text".to_string(), + ])), + ) + .await?; + +/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async fn split_by_state( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> { + let incoming: HashMap = latest_chunks + .iter() + .map(|c| (c.point_id.clone(), c.clone())) + .collect(); + + let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect(); + let points = client + .get_points( + GetPointsBuilder::new(COLLECTION, ids) + .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()])) + .with_vectors(false), + ) + .await?; + + let mut stored: HashMap = HashMap::new(); + for p in points.result { + let hash = p.get("content_hash").as_str().cloned(); + if let (Some(PointIdOptions::Uuid(id)), Some(hash)) = + (p.id.and_then(|i| i.point_id_options), hash) + { + stored.insert(id, hash); + } + } + + let (mut unchanged, mut content_changed, mut unknown_ids) = + (Vec::new(), Vec::new(), Vec::new()); + for (pid, c) in &incoming { + if stored.get(pid) == Some(&c.content_hash) { + unchanged.push(c.clone()); + } else if stored.contains_key(pid) { + content_changed.push(c.clone()); + } else { + unknown_ids.push(c.clone()); + } + } + + Ok((incoming, unchanged, content_changed, unknown_ids)) +} + +let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(&client, &latest_chunks).await?; + +async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> { + if content_changed.is_empty() { + return Ok(()); + } + let points: Vec = content_changed + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + Ok(()) +} + +/// Reuse an existing embedding when the same text is already stored; embed only what is new. +async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> { + let (mut reused, mut added) = (0, 0); + + for c in unknown_ids { + let same_text = + Filter::must([Condition::matches("content_hash", c.content_hash.clone())]); + let hits = client + .scroll( + ScrollPointsBuilder::new(COLLECTION) + .filter(same_text) + .limit(1) + .with_payload(PayloadIncludeSelector::new(vec![ + "last_updated".to_string() + ])) + .with_vectors(true), + ) + .await? + .result; + + let point = if let Some(hit) = hits.into_iter().next() { + // same text, new address: copy the vector, keep its last_updated + let last_updated = hit.get("last_updated").as_str().cloned(); + let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) { + Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector { + Some(vector_output::Vector::Dense(dense)) => dense.data, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }; + reused += 1; + PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?) + } else { + // genuinely new content: embed and insert + added += 1; + PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + ) + }; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true)) + .await?; + } + + Ok((reused, added)) +} + +/// Remove every point the current crawl no longer contains. Returns how many. +async fn delete_gone( + client: &Qdrant, + incoming_ids: &HashMap, +) -> anyhow::Result { + if incoming_ids.is_empty() { + anyhow::bail!("Refusing to delete from an empty source snapshot."); + } + + let stale = Filter::must_not([Condition::has_id( + incoming_ids.keys().map(|id| PointId::from(id.as_str())), + )]); + + let to_delete = client + .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone())) + .await? + .result + .map(|r| r.count) + .unwrap_or(0); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client + .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true)) + .await?; + Ok(to_delete) +} + +async fn sync( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result> { + check_gate(client).await?; // refuse to mix embedding models or pipeline versions + + let chunks = prepare_chunks_for_sync(latest_chunks); + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(client, &chunks).await?; + + re_embed_changed(client, &content_changed).await?; + let (reused, added) = reuse_or_add(client, &unknown_ids).await?; + let deleted = delete_gone(client, &incoming_ids).await?; + + Ok(HashMap::from([ + ("unchanged", unchanged.len()), + ("re-embedded", content_changed.len()), + ("reused_embedding", reused), + ("added", added), + ("deleted", deleted as usize), + ])) +} + +let run = sync(&client, &latest_chunks).await?; +println!("{run:?}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md new file mode 100644 index 000000000..c0097803d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md @@ -0,0 +1,10 @@ +```csharp +var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.QueryAsync( + collectionName: COLLECTION, + query: new Document { Text = QUERY, Model = MODEL }, + limit: 3, + payloadSelector: new[] { "section_url", "text" } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md new file mode 100644 index 000000000..c07149d4a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md @@ -0,0 +1,10 @@ +```go +QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: COLLECTION, + Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}), + Limit: qdrant.PtrOf(uint64(3)), + WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md new file mode 100644 index 000000000..8e0edf31b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md @@ -0,0 +1,19 @@ +```java +static final String QUERY = + "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +static void search() throws Exception { + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(COLLECTION) + .setQuery( + nearest( + Document.newBuilder() + .setText(QUERY) + .setModel(MODEL) + .build())) + .setLimit(3) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text"))) + .build()).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md new file mode 100644 index 000000000..369f79be3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md @@ -0,0 +1,10 @@ +```python +QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.query_points( + COLLECTION, + query=models.Document(text=QUERY, model=MODEL), + limit=3, + with_payload=["section_url", "text"], +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md new file mode 100644 index 000000000..4e3286158 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md @@ -0,0 +1,15 @@ +```rust +const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +client + .query( + QueryPointsBuilder::new(COLLECTION) + .query(Query::new_nearest(Document::new(QUERY, MODEL))) + .limit(3) + .with_payload(PayloadIncludeSelector::new(vec![ + "section_url".to_string(), + "text".to_string(), + ])), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md new file mode 100644 index 000000000..cfea3e665 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md @@ -0,0 +1,9 @@ +```typescript +const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.query(COLLECTION, { + query: { text: QUERY, model: MODEL }, + limit: 3, + with_payload: ["section_url", "text"], +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md new file mode 100644 index 000000000..d5236555d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md @@ -0,0 +1,35 @@ +```csharp +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)> + SplitByState(List latestChunks) +{ + var incoming = latestChunks.ToDictionary(c => c.PointId); + + var stored = new Dictionary(); + var points = await client.RetrieveAsync( + COLLECTION, + ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(), + payloadSelector: new[] { "content_hash" }, + vectorSelector: false + ); + foreach (var p in points) + stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue; + + var unchanged = new List(); + var contentChanged = new List(); + var unknownIds = new List(); + foreach (var (pid, c) in incoming) + { + if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash) + unchanged.Add(c); + else if (stored.ContainsKey(pid)) + contentChanged.Add(c); + else + unknownIds.Add(c); + } + + return (incoming, unchanged, contentChanged, unknownIds); +} + +var splitState = await SplitByState(LATEST_CHUNKS); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md new file mode 100644 index 000000000..246637600 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md @@ -0,0 +1,39 @@ +```go +// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown +splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) { + incoming := make(map[string]Chunk, len(latestChunks)) + ids := make([]*qdrant.PointId, 0, len(latestChunks)) + for _, c := range latestChunks { + incoming[c.PointID] = c + ids = append(ids, qdrant.NewID(c.PointID)) + } + + retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{ + CollectionName: COLLECTION, + Ids: ids, + WithPayload: qdrant.NewWithPayloadInclude("content_hash"), + WithVectors: qdrant.NewWithVectors(false), + }) + stored := make(map[string]string, len(retrieved)) + for _, p := range retrieved { + stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue() + } + + var unchanged, contentChanged, unknownIDs []Chunk + for pid, c := range incoming { + storedHash, found := stored[pid] + switch { + case found && storedHash == c.ContentHash: + unchanged = append(unchanged, c) + case found: + contentChanged = append(contentChanged, c) + default: + unknownIDs = append(unknownIDs, c) + } + } + + return incoming, unchanged, contentChanged, unknownIDs +} + +incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md new file mode 100644 index 000000000..439fed1f7 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md @@ -0,0 +1,43 @@ +```java +static class SyncState { + Map incoming = new LinkedHashMap<>(); + List unchanged = new ArrayList<>(); + List contentChanged = new ArrayList<>(); + List unknownIds = new ArrayList<>(); +} + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +static SyncState splitByState(List latestChunks) throws Exception { + SyncState state = new SyncState(); + for (Chunk c : latestChunks) { + state.incoming.put(c.pointId, c); + } + + Map stored = new HashMap<>(); + var points = client.retrieveAsync( + COLLECTION, + state.incoming.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()), + WithPayloadSelectorFactory.include(List.of("content_hash")), + WithVectorsSelectorFactory.enable(false), + null).get(); + for (var p : points) { + stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue()); + } + + for (Map.Entry e : state.incoming.entrySet()) { + String pid = e.getKey(); + Chunk c = e.getValue(); + if (c.contentHash.equals(stored.get(pid))) { + state.unchanged.add(c); + } else if (stored.containsKey(pid)) { + state.contentChanged.add(c); + } else { + state.unknownIds.add(c); + } + } + + return state; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md new file mode 100644 index 000000000..227950dcd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md @@ -0,0 +1,28 @@ +```python +def split_by_state(latest_chunks): + """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.""" + incoming = {c["point_id"]: c for c in latest_chunks} + + stored = {} + points = client.retrieve( + COLLECTION, + ids=list(incoming), + with_payload=["content_hash"], + with_vectors=False, + ) + for p in points: + stored[str(p.id)] = p.payload["content_hash"] + + unchanged, content_changed, unknown_ids = [], [], [] + for pid, c in incoming.items(): + if stored.get(pid) == c["content_hash"]: + unchanged.append(c) + elif pid in stored: + content_changed.append(c) + else: + unknown_ids.append(c) + + return incoming, unchanged, content_changed, unknown_ids + +incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md new file mode 100644 index 000000000..7abb1271c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md @@ -0,0 +1,48 @@ +```rust +/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async fn split_by_state( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> { + let incoming: HashMap = latest_chunks + .iter() + .map(|c| (c.point_id.clone(), c.clone())) + .collect(); + + let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect(); + let points = client + .get_points( + GetPointsBuilder::new(COLLECTION, ids) + .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()])) + .with_vectors(false), + ) + .await?; + + let mut stored: HashMap = HashMap::new(); + for p in points.result { + let hash = p.get("content_hash").as_str().cloned(); + if let (Some(PointIdOptions::Uuid(id)), Some(hash)) = + (p.id.and_then(|i| i.point_id_options), hash) + { + stored.insert(id, hash); + } + } + + let (mut unchanged, mut content_changed, mut unknown_ids) = + (Vec::new(), Vec::new(), Vec::new()); + for (pid, c) in &incoming { + if stored.get(pid) == Some(&c.content_hash) { + unchanged.push(c.clone()); + } else if stored.contains_key(pid) { + content_changed.push(c.clone()); + } else { + unknown_ids.push(c.clone()); + } + } + + Ok((incoming, unchanged, content_changed, unknown_ids)) +} + +let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(&client, &latest_chunks).await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md new file mode 100644 index 000000000..1021c8191 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md @@ -0,0 +1,33 @@ +```typescript +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async function splitByState(latestChunks: SyncChunk[]) { + const incoming = new Map(latestChunks.map((c) => [c.point_id, c])); + + const stored = new Map(); + const points = await client.retrieve(COLLECTION, { + ids: [...incoming.keys()], + with_payload: ["content_hash"], + with_vector: false, + }); + for (const p of points) { + stored.set(String(p.id), p.payload?.content_hash as string); + } + + const unchanged: SyncChunk[] = []; + const contentChanged: SyncChunk[] = []; + const unknownIds: SyncChunk[] = []; + for (const [pid, c] of incoming) { + if (stored.get(pid) === c.content_hash) { + unchanged.push(c); + } else if (stored.has(pid)) { + contentChanged.push(c); + } else { + unknownIds.push(c); + } + } + + return { incoming, unchanged, contentChanged, unknownIds }; +} + +const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md new file mode 100644 index 000000000..010dcf834 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md @@ -0,0 +1,22 @@ +```csharp +async Task> Sync(List latestChunks) +{ + await CheckGate(); // refuse to mix embedding models or pipeline versions + + var chunks = PrepareChunksForSync(latestChunks); + var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks); + + await ReEmbedChanged(contentChanged); + var (reused, added) = await ReuseOrAdd(unknownIds); + var deleted = await DeleteGone(incomingIds); + + return new Dictionary + { + ["unchanged"] = unchanged.Count, + ["re-embedded"] = contentChanged.Count, + ["reused_embedding"] = reused, + ["added"] = added, + ["deleted"] = (long)deleted, + }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md new file mode 100644 index 000000000..c352bb06a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md @@ -0,0 +1,20 @@ +```go +sync := func(latestChunks []Chunk) map[string]int { + checkGate() // refuse to mix embedding models or pipeline versions + + chunks := prepareChunksForSync(latestChunks) + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks) + + reEmbedChanged(contentChanged) + reused, added := reuseOrAdd(unknownIDs) + deleted := deleteGone(incomingIDs) + + return map[string]int{ + "unchanged": len(unchanged), + "re-embedded": len(contentChanged), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md new file mode 100644 index 000000000..4060b01ba --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md @@ -0,0 +1,19 @@ +```java +static Map sync(List latestChunks) throws Exception { + checkGate(); // refuse to mix embedding models or pipeline versions + + List chunks = prepareChunksForSync(latestChunks); + SyncState state = splitByState(chunks); + + reEmbedChanged(state.contentChanged); + int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added} + long deleted = deleteGone(state.incoming); + + return Map.of( + "unchanged", (long) state.unchanged.size(), + "re-embedded", (long) state.contentChanged.size(), + "reused_embedding", (long) reusedAdded[0], + "added", (long) reusedAdded[1], + "deleted", deleted); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md new file mode 100644 index 000000000..568186056 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md @@ -0,0 +1,19 @@ +```python +def sync(latest_chunks): + check_gate() # refuse to mix embedding models or pipeline versions + + chunks = prepare_chunks_for_sync(latest_chunks) + incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks) + + re_embed_changed(content_changed) + reused, added = reuse_or_add(unknown_ids) + deleted = delete_gone(incoming_ids) + + return { + "unchanged": len(unchanged), + "re-embedded": len(content_changed), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md new file mode 100644 index 000000000..098e594d0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md @@ -0,0 +1,24 @@ +```rust +async fn sync( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result> { + check_gate(client).await?; // refuse to mix embedding models or pipeline versions + + let chunks = prepare_chunks_for_sync(latest_chunks); + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(client, &chunks).await?; + + re_embed_changed(client, &content_changed).await?; + let (reused, added) = reuse_or_add(client, &unknown_ids).await?; + let deleted = delete_gone(client, &incoming_ids).await?; + + Ok(HashMap::from([ + ("unchanged", unchanged.len()), + ("re-embedded", content_changed.len()), + ("reused_embedding", reused), + ("added", added), + ("deleted", deleted as usize), + ])) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md new file mode 100644 index 000000000..95662c80a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md @@ -0,0 +1,20 @@ +```typescript +async function sync(latestChunks: RawChunk[]) { + await checkGate(); // refuse to mix embedding models or pipeline versions + + const chunks = prepareChunksForSync(latestChunks); + const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks); + + await reEmbedChanged(contentChanged); + const { reused, added } = await reuseOrAdd(unknownIds); + const deleted = await deleteGone(incoming); + + return { + "unchanged": unchanged.length, + "re-embedded": contentChanged.length, + "reused_embedding": reused, + "added": added, + "deleted": deleted, + }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md new file mode 100644 index 000000000..17c9296fd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md @@ -0,0 +1,230 @@ +```typescript +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +const QDRANT_URL = process.env.QDRANT_URL; +const QDRANT_API_KEY = process.env.QDRANT_API_KEY; + +const client = new QdrantClient({ + url: QDRANT_URL, + apiKey: QDRANT_API_KEY, +}); + +const MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE = "docs-prep-pipeline-v1"; +const COLLECTION = "docs-sync-tutorial"; + +await client.createCollection(COLLECTION, { + vectors: { + size: 384, // all-MiniLM-L6-v2 output dimension + distance: "Cosine", + }, +}); + +await client.updateCollection(COLLECTION, { + metadata: { embedding_model: MODEL, pipeline_version: PIPELINE }, +}); + +async function checkGate() { + // compare this pipeline's constants against what the collection records about itself + const meta = ((await client.getCollection(COLLECTION)).config.metadata ?? + {}) as Record; + + if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) { + throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`); + } +} + +import { createHash } from "node:crypto"; + +type RawChunk = { url: string; anchor: string; chunk_num: number; text: string }; +type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string }; + +function contentHash(text: string): string { + return createHash("sha256").update(text).digest("hex"); +} + +// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name +function pointId(url: string, anchor: string, num: number): string { + // Qdrant accepts any well-formed UUID as a point ID: + // hash the address, format the digest as a UUID, and the same address always yields the same ID + const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex"); + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`; +} + +// Derive both values (and the section address) for every raw chunk. +function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] { + return chunks.map((c) => { + const text = normalize(c.text); + return { + ...c, + text, + section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url, + content_hash: contentHash(text), + point_id: pointId(c.url, c.anchor, c.chunk_num), + }; + }); +} + +function payload(chunk: SyncChunk, lastUpdated?: string) { + return { + url: chunk.url, + anchor: chunk.anchor, + chunk_num: chunk.chunk_num, + section_url: chunk.section_url, + text: chunk.text, + content_hash: chunk.content_hash, + last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"), + }; +} + +for (const field of ["content_hash", "url", "section_url"]) { + await client.createPayloadIndex(COLLECTION, { + field_name: field, + field_schema: "keyword", + }); +} + +await client.upsert(COLLECTION, { + points: prepareChunksForSync(CHUNKS).map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, +}); + +const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.query(COLLECTION, { + query: { text: QUERY, model: MODEL }, + limit: 3, + with_payload: ["section_url", "text"], +}); + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async function splitByState(latestChunks: SyncChunk[]) { + const incoming = new Map(latestChunks.map((c) => [c.point_id, c])); + + const stored = new Map(); + const points = await client.retrieve(COLLECTION, { + ids: [...incoming.keys()], + with_payload: ["content_hash"], + with_vector: false, + }); + for (const p of points) { + stored.set(String(p.id), p.payload?.content_hash as string); + } + + const unchanged: SyncChunk[] = []; + const contentChanged: SyncChunk[] = []; + const unknownIds: SyncChunk[] = []; + for (const [pid, c] of incoming) { + if (stored.get(pid) === c.content_hash) { + unchanged.push(c); + } else if (stored.has(pid)) { + contentChanged.push(c); + } else { + unknownIds.push(c); + } + } + + return { incoming, unchanged, contentChanged, unknownIds }; +} + +const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS); + +async function reEmbedChanged(contentChanged: SyncChunk[]) { + if (contentChanged.length === 0) { + return; + } + await client.upsert(COLLECTION, { + points: contentChanged.map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, + }); +} + +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async function reuseOrAdd(unknownIds: SyncChunk[]) { + let reused = 0; + let added = 0; + + for (const c of unknownIds) { + const sameText = { + must: [ + { + key: "content_hash", + match: { value: c.content_hash }, + }, + ], + }; + const hits = (await client.scroll(COLLECTION, { + filter: sameText, + limit: 1, + with_payload: ["last_updated"], + with_vector: true, + })).points; + + let point: Schemas["PointStruct"]; + if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated + point = { + id: c.point_id, + vector: hits[0].vector as number[], + payload: payload(c, hits[0].payload?.last_updated as string), + }; + reused += 1; + } else { // genuinely new content: embed and insert + point = { + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + }; + added += 1; + } + + await client.upsert(COLLECTION, { points: [point], wait: true }); + } + + return { reused, added }; +} + +// Remove every point the current crawl no longer contains. Returns how many. +async function deleteGone(incoming: Map) { + if (incoming.size === 0) { + throw new Error("Refusing to delete from an empty source snapshot."); + } + + const stale = { must_not: [{ has_id: [...incoming.keys()] }] }; + + const toDelete = (await client.count(COLLECTION, { filter: stale })).count; + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.delete(COLLECTION, { filter: stale, wait: true }); + return toDelete; +} + +async function sync(latestChunks: RawChunk[]) { + await checkGate(); // refuse to mix embedding models or pipeline versions + + const chunks = prepareChunksForSync(latestChunks); + const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks); + + await reEmbedChanged(contentChanged); + const { reused, added } = await reuseOrAdd(unknownIds); + const deleted = await deleteGone(incoming); + + return { + "unchanged": unchanged.length, + "re-embedded": contentChanged.length, + "reused_embedding": reused, + "added": added, + "deleted": deleted, + }; +} + +const run = await sync(LATEST_CHUNKS); +console.log(run); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go new file mode 100644 index 000000000..ff8a35d3e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go @@ -0,0 +1,359 @@ +package snippet + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "os" + "regexp" + "strings" + "time" + + "github.com/google/uuid" + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @block-start client-connection + QDRANT_URL := os.Getenv("QDRANT_URL") + QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY") + + client, err := qdrant.NewClient(&qdrant.Config{ + Host: QDRANT_URL, + APIKey: QDRANT_API_KEY, + UseTLS: true, + }) + // @block-end client-connection + + // @hide-start + if err != nil { + panic(err) + } + + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and normalize() live in the tutorial notebook + type Chunk struct { + URL string + Anchor string + ChunkNum int + Text string + SectionURL string + ContentHash string + PointID string + } + + CHUNKS := []Chunk{ + { + URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "prerequisites", + ChunkNum: 0, + Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + }, + { + URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "step-3-enable-an-admin-api-key", + ChunkNum: 0, + Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + }, + } + + invisibleChars := regexp.MustCompile("[\u200B\u200C\u200D\uFEFF\u00AD]") // zero-width chars and soft hyphen + whitespace := regexp.MustCompile(`\s+`) + normalize := func(text string) string { + text = invisibleChars.ReplaceAllString(text, "") + return strings.TrimSpace(whitespace.ReplaceAllString(text, " ")) + } + // @hide-end + + // @block-start create-collection + MODEL := "sentence-transformers/all-MiniLM-L6-v2" + PIPELINE := "docs-prep-pipeline-v1" + COLLECTION := "docs-sync-tutorial" + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: COLLECTION, + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 384, // all-MiniLM-L6-v2 output dimension + Distance: qdrant.Distance_Cosine, + }), + Metadata: qdrant.NewValueMap(map[string]any{ + "embedding_model": MODEL, + "pipeline_version": PIPELINE, + }), + }) + // @block-end create-collection + + // @block-start check-gate + checkGate := func() { + // compare this pipeline's constants against what the collection records about itself + info, err := client.GetCollectionInfo(context.Background(), COLLECTION) + if err != nil { panic(err) } // @hide + meta := info.GetConfig().GetMetadata() + + if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE { + panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta)) + } + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + contentHash := func(text string) string { + sum := sha256.Sum256([]byte(text)) + return hex.EncodeToString(sum[:]) + } + + pointID := func(url, anchor string, num int) string { + // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires, + // marking the input as a URL-like name + return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String() + } + + // derive both values (and the section address) for every raw chunk + prepareChunksForSync := func(chunks []Chunk) []Chunk { + out := make([]Chunk, 0, len(chunks)) + for _, c := range chunks { + c.Text = normalize(c.Text) + c.SectionURL = c.URL + if c.Anchor != "" { + c.SectionURL = c.URL + "#" + c.Anchor + } + c.ContentHash = contentHash(c.Text) + c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum) + out = append(out, c) + } + return out + } + // @block-end identity-and-fingerprint + + // @block-start payload + payload := func(c Chunk, lastUpdated string) map[string]any { + if lastUpdated == "" { + lastUpdated = time.Now().UTC().Format(time.RFC3339) + } + return map[string]any{ + "url": c.URL, + "anchor": c.Anchor, + "chunk_num": c.ChunkNum, + "section_url": c.SectionURL, + "text": c.Text, + "content_hash": c.ContentHash, + "last_updated": lastUpdated, + } + } + // @block-end payload + + // @block-start payload-indexes + for _, field := range []string{"content_hash", "url", "section_url"} { + client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: COLLECTION, + FieldName: field, + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + }) + } + // @block-end payload-indexes + + // @block-start populate + var points []*qdrant.PointStruct + for _, c := range prepareChunksForSync(CHUNKS) { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) + // @block-end populate + + // @block-start search + QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + + client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: COLLECTION, + Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}), + Limit: qdrant.PtrOf(uint64(3)), + WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"), + }) + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + LATEST_CHUNKS := prepareChunksForSync(CHUNKS) + // @hide-end + + // @block-start split-by-state + // compare the incoming chunk list to the collection: who is unchanged, changed, or unknown + splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) { + incoming := make(map[string]Chunk, len(latestChunks)) + ids := make([]*qdrant.PointId, 0, len(latestChunks)) + for _, c := range latestChunks { + incoming[c.PointID] = c + ids = append(ids, qdrant.NewID(c.PointID)) + } + + retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{ + CollectionName: COLLECTION, + Ids: ids, + WithPayload: qdrant.NewWithPayloadInclude("content_hash"), + WithVectors: qdrant.NewWithVectors(false), + }) + if err != nil { panic(err) } // @hide + stored := make(map[string]string, len(retrieved)) + for _, p := range retrieved { + stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue() + } + + var unchanged, contentChanged, unknownIDs []Chunk + for pid, c := range incoming { + storedHash, found := stored[pid] + switch { + case found && storedHash == c.ContentHash: + unchanged = append(unchanged, c) + case found: + contentChanged = append(contentChanged, c) + default: + unknownIDs = append(unknownIDs, c) + } + } + + return incoming, unchanged, contentChanged, unknownIDs + } + + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS) + // @block-end split-by-state + + // @hide-start + _, _, _, _ = incomingIDs, unchanged, contentChanged, unknownIDs + // @hide-end + + // @block-start re-embed-changed + reEmbedChanged := func(contentChanged []Chunk) { + if len(contentChanged) == 0 { + return + } + points := make([]*qdrant.PointStruct, 0, len(contentChanged)) + for _, c := range contentChanged { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + // reuse an existing embedding when the same text is already stored; embed only what is new + reuseOrAdd := func(unknownIDs []Chunk) (int, int) { + reused, added := 0, 0 + + for _, c := range unknownIDs { + sameText := &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("content_hash", c.ContentHash), + }, + } + hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: COLLECTION, + Filter: sameText, + Limit: qdrant.PtrOf(uint32(1)), + WithPayload: qdrant.NewWithPayloadInclude("last_updated"), + WithVectors: qdrant.NewWithVectors(true), + }) + if err != nil { panic(err) } // @hide + + var point *qdrant.PointStruct + if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...), + Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())), + } + reused++ + } else { // genuinely new content: embed and insert + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + } + added++ + } + + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: []*qdrant.PointStruct{point}, + Wait: qdrant.PtrOf(true), + }) + } + + return reused, added + } + // @block-end reuse-or-add + + // @block-start delete-gone + // remove every point the current crawl no longer contains, return how many + deleteGone := func(incomingIDs map[string]Chunk) int { + if len(incomingIDs) == 0 { + panic("Refusing to delete from an empty source snapshot.") + } + + ids := make([]*qdrant.PointId, 0, len(incomingIDs)) + for pid := range incomingIDs { + ids = append(ids, qdrant.NewID(pid)) + } + stale := &qdrant.Filter{ + MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)}, + } + + toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{ + CollectionName: COLLECTION, + Filter: stale, + }) + if err != nil { panic(err) } // @hide + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.Delete(context.Background(), &qdrant.DeletePoints{ + CollectionName: COLLECTION, + Points: qdrant.NewPointsSelectorFilter(stale), + Wait: qdrant.PtrOf(true), + }) + return int(toDelete) + } + // @block-end delete-gone + + // @block-start sync + sync := func(latestChunks []Chunk) map[string]int { + checkGate() // refuse to mix embedding models or pipeline versions + + chunks := prepareChunksForSync(latestChunks) + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks) + + reEmbedChanged(contentChanged) + reused, added := reuseOrAdd(unknownIDs) + deleted := deleteGone(incomingIDs) + + return map[string]int{ + "unchanged": len(unchanged), + "re-embedded": len(contentChanged), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } + } + // @block-end sync + + // @block-start run-sync + run := sync(LATEST_CHUNKS) + fmt.Println(run) + // @block-end run-sync +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java new file mode 100644 index 000000000..8fdd4ee1d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java @@ -0,0 +1,415 @@ +package com.example.snippets_amalgamation; + +import static io.qdrant.client.ConditionFactory.hasId; +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.PointIdFactory.id; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorFactory.vector; +import static io.qdrant.client.VectorsFactory.vectors; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.VectorOutputHelper; +import io.qdrant.client.WithPayloadSelectorFactory; +import io.qdrant.client.WithVectorsSelectorFactory; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.JsonWithInt.Value; +import io.qdrant.client.grpc.Points.Document; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ScrollPoints; +import java.math.BigInteger; +import java.nio.charset.StandardCharsets; +import java.security.MessageDigest; +import java.time.OffsetDateTime; +import java.time.ZoneOffset; +import java.time.temporal.ChronoUnit; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import java.util.UUID; +import java.util.stream.Collectors; + +public class Snippet { + + // @block-start client-connection + static final String QDRANT_URL = System.getenv("QDRANT_URL"); + static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY"); + + static final QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true) + .withApiKey(QDRANT_API_KEY) + .build()); + // @block-end client-connection + + // @hide-start + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and normalize() live in the tutorial notebook + static class Chunk { + String url; + String anchor; + int chunkNum; + String text; + String sectionUrl; // derived in prepareChunksForSync + String contentHash; // derived in prepareChunksForSync + String pointId; // derived in prepareChunksForSync + + Chunk(String url, String anchor, int chunkNum, String text) { + this.url = url; + this.anchor = anchor; + this.chunkNum = chunkNum; + this.text = text; + } + } + + static final List CHUNKS = List.of( + new Chunk( + "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "prerequisites", + 0, + "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ..."), + new Chunk( + "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "step-3-enable-an-admin-api-key", + 0, + "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...")); + + static String normalize(String text) { + return text.replaceAll("\\s+", " ").strip(); + } + // @hide-end + + // @block-start create-collection + static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2"; + static final String PIPELINE = "docs-prep-pipeline-v1"; + static final String COLLECTION = "docs-sync-tutorial"; + + static void createCollection() throws Exception { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(COLLECTION) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(384) // all-MiniLM-L6-v2 output dimension + .setDistance(Distance.Cosine) + .build()) + .build()) + .putAllMetadata( + Map.of( + "embedding_model", value(MODEL), + "pipeline_version", value(PIPELINE))) + .build()).get(); + } + // @block-end create-collection + + // @block-start check-gate + static void checkGate() throws Exception { + // compare this pipeline's constants against what the collection records about itself + Map meta = + client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap(); + + Value model = meta.get("embedding_model"); + Value pipeline = meta.get("pipeline_version"); + if (model == null || !MODEL.equals(model.getStringValue()) + || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) { + throw new RuntimeException( + "collection was built by " + meta + ": full re-embed into a fresh collection required"); + } + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + static String contentHash(String text) throws Exception { + byte[] digest = MessageDigest.getInstance("SHA-256") + .digest(text.getBytes(StandardCharsets.UTF_8)); + return String.format("%064x", new BigInteger(1, digest)); + } + + static String pointId(String url, String anchor, int num) { + // name-based UUID (version 3); the same address always yields the same ID + return UUID.nameUUIDFromBytes( + (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString(); + } + + // Derive both values (and the section address) for every raw chunk. + static List prepareChunksForSync(List chunks) throws Exception { + List out = new ArrayList<>(); + for (Chunk c : chunks) { + String text = normalize(c.text); + Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text); + prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url; + prepared.contentHash = contentHash(text); + prepared.pointId = pointId(c.url, c.anchor, c.chunkNum); + out.add(prepared); + } + return out; + } + // @block-end identity-and-fingerprint + + // @block-start payload + static Map payload(Chunk chunk, String lastUpdated) { + Map p = new HashMap<>(); + p.put("url", value(chunk.url)); + p.put("anchor", value(chunk.anchor)); + p.put("chunk_num", value(chunk.chunkNum)); + p.put("section_url", value(chunk.sectionUrl)); + p.put("text", value(chunk.text)); + p.put("content_hash", value(chunk.contentHash)); + p.put("last_updated", value(lastUpdated != null + ? lastUpdated + : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString())); + return p; + } + // @block-end payload + + // @block-start payload-indexes + static void createPayloadIndexes() throws Exception { + for (String field : List.of("content_hash", "url", "section_url")) { + client.createPayloadIndexAsync( + COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get(); + } + } + // @block-end payload-indexes + + // @block-start populate + static void populate() throws Exception { + List points = new ArrayList<>(); + for (Chunk c : prepareChunksForSync(CHUNKS)) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); + } + // @block-end populate + + // @block-start search + static final String QUERY = + "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + + static void search() throws Exception { + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(COLLECTION) + .setQuery( + nearest( + Document.newBuilder() + .setText(QUERY) + .setModel(MODEL) + .build())) + .setLimit(3) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text"))) + .build()).get(); + } + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + static List LATEST_CHUNKS; + // @hide-end + + // @block-start split-by-state + static class SyncState { + Map incoming = new LinkedHashMap<>(); + List unchanged = new ArrayList<>(); + List contentChanged = new ArrayList<>(); + List unknownIds = new ArrayList<>(); + } + + // Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. + static SyncState splitByState(List latestChunks) throws Exception { + SyncState state = new SyncState(); + for (Chunk c : latestChunks) { + state.incoming.put(c.pointId, c); + } + + Map stored = new HashMap<>(); + var points = client.retrieveAsync( + COLLECTION, + state.incoming.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()), + WithPayloadSelectorFactory.include(List.of("content_hash")), + WithVectorsSelectorFactory.enable(false), + null).get(); + for (var p : points) { + stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue()); + } + + for (Map.Entry e : state.incoming.entrySet()) { + String pid = e.getKey(); + Chunk c = e.getValue(); + if (c.contentHash.equals(stored.get(pid))) { + state.unchanged.add(c); + } else if (stored.containsKey(pid)) { + state.contentChanged.add(c); + } else { + state.unknownIds.add(c); + } + } + + return state; + } + // @block-end split-by-state + + // @block-start re-embed-changed + static void reEmbedChanged(List contentChanged) throws Exception { + if (contentChanged.isEmpty()) { + return; + } + List points = new ArrayList<>(); + for (Chunk c : contentChanged) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + // Reuse an existing embedding when the same text is already stored; embed only what is new. + static int[] reuseOrAdd(List unknownIds) throws Exception { + int reused = 0; + int added = 0; + + for (Chunk c : unknownIds) { + Filter sameText = Filter.newBuilder() + .addMust(matchKeyword("content_hash", c.contentHash)) + .build(); + + var hits = client.scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName(COLLECTION) + .setFilter(sameText) + .setLimit(1) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated"))) + .setWithVectors(WithVectorsSelectorFactory.enable(true)) + .build()).get().getResultList(); + + PointStruct point; + if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors(vectors(vector( + VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector()) + .getDataList()))) + .putAllPayload( + payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue())) + .build(); + reused++; + } else { // genuinely new content: embed and insert + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build(); + added++; + } + + client.upsertAsync(COLLECTION, List.of(point)).get(); + } + + return new int[] {reused, added}; + } + // @block-end reuse-or-add + + // @block-start delete-gone + // Remove every point the current crawl no longer contains. Returns how many. + static long deleteGone(Map incomingIds) throws Exception { + if (incomingIds.isEmpty()) { + throw new IllegalArgumentException("Refusing to delete from an empty source snapshot."); + } + + Filter stale = Filter.newBuilder() + .addMustNot(hasId( + incomingIds.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()))) + .build(); + + long toDelete = client.countAsync(COLLECTION, stale, true).get(); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.deleteAsync(COLLECTION, stale).get(); + return toDelete; + } + // @block-end delete-gone + + // @block-start sync + static Map sync(List latestChunks) throws Exception { + checkGate(); // refuse to mix embedding models or pipeline versions + + List chunks = prepareChunksForSync(latestChunks); + SyncState state = splitByState(chunks); + + reEmbedChanged(state.contentChanged); + int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added} + long deleted = deleteGone(state.incoming); + + return Map.of( + "unchanged", (long) state.unchanged.size(), + "re-embedded", (long) state.contentChanged.size(), + "reused_embedding", (long) reusedAdded[0], + "added", (long) reusedAdded[1], + "deleted", deleted); + } + // @block-end sync + + // @block-start run-sync + static void runSync() throws Exception { + Map run = sync(LATEST_CHUNKS); + System.out.println(run); + } + // @block-end run-sync + + // @hide-start + public static void run() throws Exception { + createCollection(); + createPayloadIndexes(); + populate(); + search(); + + LATEST_CHUNKS = prepareChunksForSync(CHUNKS); + SyncState state = splitByState(LATEST_CHUNKS); + + runSync(); + // @hide-end + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py new file mode 100644 index 000000000..346e6f106 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py @@ -0,0 +1,261 @@ +# @block-start client-connection +import os + +from qdrant_client import QdrantClient, models + +QDRANT_URL = os.getenv("QDRANT_URL") +QDRANT_API_KEY = os.getenv("QDRANT_API_KEY") + +client = QdrantClient( + url=QDRANT_URL, + api_key=QDRANT_API_KEY, + cloud_inference=True +) +# @block-end client-connection + +# @hide-start +# data and text normalization are not the lesson of this tutorial: +# the full CHUNKS list and normalize() live in the tutorial notebook +CHUNKS = [ + { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "prerequisites", + "chunk_num": 0, + "text": "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + }, + { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "step-3-enable-an-admin-api-key", + "chunk_num": 0, + "text": "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + }, +] + +import re +import unicodedata + +def normalize(text): + text = unicodedata.normalize("NFKC", text) + text = text.translate(dict.fromkeys(map(ord, "​‌‍­"))) + return re.sub(r"\s+", " ", text).strip() +# @hide-end + +# @block-start create-collection +MODEL = "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE = "docs-prep-pipeline-v1" +COLLECTION = "docs-sync-tutorial" + +client.create_collection( + COLLECTION, + vectors_config=models.VectorParams( + size=384, # all-MiniLM-L6-v2 output dimension + distance=models.Distance.COSINE, + ), + metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE}, +) +# @block-end create-collection + +# @block-start check-gate +def check_gate(): + # compare this pipeline's constants against what the collection records about itself + meta = client.get_collection(COLLECTION).config.metadata or {} + + if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE: + raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required") +# @block-end check-gate + +# @block-start identity-and-fingerprint +import hashlib +import uuid +from datetime import datetime, timezone + +def content_hash(text): + return hashlib.sha256(text.encode()).hexdigest() + +def point_id(url, anchor, num): + # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}")) + +def prepare_chunks_for_sync(chunks): + """Derive both values (and the section address) for every raw chunk.""" + out = [] + for c in chunks: + text = normalize(c["text"]) + out.append({ + **c, + "text": text, + "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"], + "content_hash": content_hash(text), + "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]), + }) + return out +# @block-end identity-and-fingerprint + +# @block-start payload +def payload(chunk, last_updated=None): + return { + "url": chunk["url"], + "anchor": chunk["anchor"], + "chunk_num": chunk["chunk_num"], + "section_url": chunk["section_url"], + "text": chunk["text"], + "content_hash": chunk["content_hash"], + "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"), + } +# @block-end payload + +# @block-start payload-indexes +for field in ("content_hash", "url", "section_url"): + client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD) +# @block-end payload-indexes + +# @block-start populate +client.upsert(COLLECTION, points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in prepare_chunks_for_sync(CHUNKS) +], wait=True) +# @block-end populate + +# @block-start search +QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.query_points( + COLLECTION, + query=models.Document(text=QUERY, model=MODEL), + limit=3, + with_payload=["section_url", "text"], +) +# @block-end search + +# @hide-start +# the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook +LATEST_CHUNKS = prepare_chunks_for_sync(CHUNKS) +# @hide-end + +# @block-start split-by-state +def split_by_state(latest_chunks): + """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.""" + incoming = {c["point_id"]: c for c in latest_chunks} + + stored = {} + points = client.retrieve( + COLLECTION, + ids=list(incoming), + with_payload=["content_hash"], + with_vectors=False, + ) + for p in points: + stored[str(p.id)] = p.payload["content_hash"] + + unchanged, content_changed, unknown_ids = [], [], [] + for pid, c in incoming.items(): + if stored.get(pid) == c["content_hash"]: + unchanged.append(c) + elif pid in stored: + content_changed.append(c) + else: + unknown_ids.append(c) + + return incoming, unchanged, content_changed, unknown_ids + +incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS) +# @block-end split-by-state + +# @block-start re-embed-changed +def re_embed_changed(content_changed): + if not content_changed: + return + client.upsert(COLLECTION, + points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in content_changed], + wait=True) +# @block-end re-embed-changed + +# @block-start reuse-or-add +def reuse_or_add(unknown_ids): + """Reuse an existing embedding when the same text is already stored; embed only what is new.""" + reused, added = 0, 0 + + for c in unknown_ids: + same_text = models.Filter(must=[ + models.FieldCondition( + key="content_hash", + match=models.MatchValue(value=c["content_hash"]), + ) + ]) + hits, _ = client.scroll( + COLLECTION, + scroll_filter=same_text, + limit=1, + with_payload=["last_updated"], + with_vectors=True, + ) + + if hits: # same text, new address: copy the vector, keep its last_updated + point = models.PointStruct( + id=c["point_id"], + vector=hits[0].vector, + payload=payload(c, hits[0].payload["last_updated"]), + ) + reused += 1 + else: # genuinely new content: embed and insert + point = models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + added += 1 + + client.upsert(COLLECTION, points=[point], wait=True) + + return reused, added +# @block-end reuse-or-add + +# @block-start delete-gone +def delete_gone(incoming_ids): + """Remove every point the current crawl no longer contains. Returns how many.""" + if not incoming_ids: + raise ValueError("Refusing to delete from an empty source snapshot.") + + stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))]) + + to_delete = client.count(COLLECTION, count_filter=stale).count + + # potential check against a threshold to avoid accidental mass deletion could be added here + client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True) + return to_delete +# @block-end delete-gone + +# @block-start sync +def sync(latest_chunks): + check_gate() # refuse to mix embedding models or pipeline versions + + chunks = prepare_chunks_for_sync(latest_chunks) + incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks) + + re_embed_changed(content_changed) + reused, added = reuse_or_add(unknown_ids) + deleted = delete_gone(incoming_ids) + + return { + "unchanged": len(unchanged), + "re-embedded": len(content_changed), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +# @block-end sync + +# @block-start run-sync +run = sync(LATEST_CHUNKS) +print(run) +# @block-end run-sync diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs new file mode 100644 index 000000000..fcc22bf83 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs @@ -0,0 +1,397 @@ +use serde_json::{json, Value}; +use std::collections::HashMap; + +use qdrant_client::qdrant::{ + point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder, + CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance, + Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct, + Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder, +}; +use qdrant_client::{Payload, Qdrant}; +use sha2::{Digest, Sha256}; + +pub async fn main() -> anyhow::Result<()> { + // @block-start client-connection + let qdrant_url = std::env::var("QDRANT_URL")?; + let qdrant_api_key = std::env::var("QDRANT_API_KEY")?; + + let client = Qdrant::from_url(&qdrant_url) + .api_key(qdrant_api_key) + .build()?; + // @block-end client-connection + + // @hide-start + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and normalize() live in the tutorial notebook + #[derive(Clone, Default)] + struct Chunk { + url: String, + anchor: String, + chunk_num: u32, + text: String, + section_url: String, + content_hash: String, + point_id: String, + } + + let chunks: Vec = vec![ + Chunk { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(), + anchor: "prerequisites".into(), + chunk_num: 0, + text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...".into(), + ..Default::default() + }, + Chunk { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(), + anchor: "step-3-enable-an-admin-api-key".into(), + chunk_num: 0, + text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...".into(), + ..Default::default() + }, + ]; + + fn normalize(text: &str) -> String { + text.split_whitespace().collect::>().join(" ") + } + // @hide-end + + // @block-start create-collection + const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2"; + const PIPELINE: &str = "docs-prep-pipeline-v1"; + const COLLECTION: &str = "docs-sync-tutorial"; + + let mut metadata: HashMap = HashMap::new(); + metadata.insert("embedding_model".to_string(), json!(MODEL)); + metadata.insert("pipeline_version".to_string(), json!(PIPELINE)); + + client + .create_collection( + CreateCollectionBuilder::new(COLLECTION) + .vectors_config(VectorParamsBuilder::new( + 384, // all-MiniLM-L6-v2 output dimension + Distance::Cosine, + )) + .metadata(metadata), + ) + .await?; + // @block-end create-collection + + // @block-start check-gate + async fn check_gate(client: &Qdrant) -> anyhow::Result<()> { + // compare this pipeline's constants against what the collection records about itself + let meta = client + .collection_info(COLLECTION) + .await? + .result + .and_then(|info| info.config) + .map(|config| config.metadata) + .unwrap_or_default(); + + if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL) + || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str) + != Some(PIPELINE) + { + anyhow::bail!( + "collection was built by {meta:?}: full re-embed into a fresh collection required" + ); + } + Ok(()) + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + fn content_hash(text: &str) -> String { + Sha256::digest(text.as_bytes()) + .iter() + .map(|byte| format!("{byte:02x}")) + .collect() + } + + fn point_id(url: &str, anchor: &str, num: u32) -> String { + // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + uuid::Uuid::new_v5( + &uuid::Uuid::NAMESPACE_URL, + format!("{url}#{anchor}::{num}").as_bytes(), + ) + .to_string() + } + + /// Derive both values (and the section address) for every raw chunk. + fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec { + chunks + .iter() + .map(|c| { + let text = normalize(&c.text); + Chunk { + text: text.clone(), + section_url: if c.anchor.is_empty() { + c.url.clone() + } else { + format!("{}#{}", c.url, c.anchor) + }, + content_hash: content_hash(&text), + point_id: point_id(&c.url, &c.anchor, c.chunk_num), + ..c.clone() + } + }) + .collect() + } + // @block-end identity-and-fingerprint + + // @block-start payload + fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result { + let last_updated = last_updated.unwrap_or_else(|| { + chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false) + }); + Ok(Payload::try_from(serde_json::json!({ + "url": chunk.url, + "anchor": chunk.anchor, + "chunk_num": chunk.chunk_num, + "section_url": chunk.section_url, + "text": chunk.text, + "content_hash": chunk.content_hash, + "last_updated": last_updated, + }))?) + } + // @block-end payload + + // @block-start payload-indexes + for field in ["content_hash", "url", "section_url"] { + client + .create_field_index(CreateFieldIndexCollectionBuilder::new( + COLLECTION, + field, + FieldType::Keyword, + )) + .await?; + } + // @block-end payload-indexes + + // @block-start populate + let points: Vec = prepare_chunks_for_sync(&chunks) + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + // @block-end populate + + // @block-start search + const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + + client + .query( + QueryPointsBuilder::new(COLLECTION) + .query(Query::new_nearest(Document::new(QUERY, MODEL))) + .limit(3) + .with_payload(PayloadIncludeSelector::new(vec![ + "section_url".to_string(), + "text".to_string(), + ])), + ) + .await?; + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + let latest_chunks = prepare_chunks_for_sync(&chunks); + // @hide-end + + // @block-start split-by-state + /// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. + async fn split_by_state( + client: &Qdrant, + latest_chunks: &[Chunk], + ) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> { + let incoming: HashMap = latest_chunks + .iter() + .map(|c| (c.point_id.clone(), c.clone())) + .collect(); + + let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect(); + let points = client + .get_points( + GetPointsBuilder::new(COLLECTION, ids) + .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()])) + .with_vectors(false), + ) + .await?; + + let mut stored: HashMap = HashMap::new(); + for p in points.result { + let hash = p.get("content_hash").as_str().cloned(); + if let (Some(PointIdOptions::Uuid(id)), Some(hash)) = + (p.id.and_then(|i| i.point_id_options), hash) + { + stored.insert(id, hash); + } + } + + let (mut unchanged, mut content_changed, mut unknown_ids) = + (Vec::new(), Vec::new(), Vec::new()); + for (pid, c) in &incoming { + if stored.get(pid) == Some(&c.content_hash) { + unchanged.push(c.clone()); + } else if stored.contains_key(pid) { + content_changed.push(c.clone()); + } else { + unknown_ids.push(c.clone()); + } + } + + Ok((incoming, unchanged, content_changed, unknown_ids)) + } + + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(&client, &latest_chunks).await?; + // @block-end split-by-state + + // @hide-start + _ = (&incoming_ids, &unchanged, &content_changed, &unknown_ids); + // @hide-end + + // @block-start re-embed-changed + async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> { + if content_changed.is_empty() { + return Ok(()); + } + let points: Vec = content_changed + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + Ok(()) + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + /// Reuse an existing embedding when the same text is already stored; embed only what is new. + async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> { + let (mut reused, mut added) = (0, 0); + + for c in unknown_ids { + let same_text = + Filter::must([Condition::matches("content_hash", c.content_hash.clone())]); + let hits = client + .scroll( + ScrollPointsBuilder::new(COLLECTION) + .filter(same_text) + .limit(1) + .with_payload(PayloadIncludeSelector::new(vec![ + "last_updated".to_string() + ])) + .with_vectors(true), + ) + .await? + .result; + + let point = if let Some(hit) = hits.into_iter().next() { + // same text, new address: copy the vector, keep its last_updated + let last_updated = hit.get("last_updated").as_str().cloned(); + let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) { + Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector { + Some(vector_output::Vector::Dense(dense)) => dense.data, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }; + reused += 1; + PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?) + } else { + // genuinely new content: embed and insert + added += 1; + PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + ) + }; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true)) + .await?; + } + + Ok((reused, added)) + } + // @block-end reuse-or-add + + // @block-start delete-gone + /// Remove every point the current crawl no longer contains. Returns how many. + async fn delete_gone( + client: &Qdrant, + incoming_ids: &HashMap, + ) -> anyhow::Result { + if incoming_ids.is_empty() { + anyhow::bail!("Refusing to delete from an empty source snapshot."); + } + + let stale = Filter::must_not([Condition::has_id( + incoming_ids.keys().map(|id| PointId::from(id.as_str())), + )]); + + let to_delete = client + .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone())) + .await? + .result + .map(|r| r.count) + .unwrap_or(0); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client + .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true)) + .await?; + Ok(to_delete) + } + // @block-end delete-gone + + // @block-start sync + async fn sync( + client: &Qdrant, + latest_chunks: &[Chunk], + ) -> anyhow::Result> { + check_gate(client).await?; // refuse to mix embedding models or pipeline versions + + let chunks = prepare_chunks_for_sync(latest_chunks); + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(client, &chunks).await?; + + re_embed_changed(client, &content_changed).await?; + let (reused, added) = reuse_or_add(client, &unknown_ids).await?; + let deleted = delete_gone(client, &incoming_ids).await?; + + Ok(HashMap::from([ + ("unchanged", unchanged.len()), + ("re-embedded", content_changed.len()), + ("reused_embedding", reused), + ("added", added), + ("deleted", deleted as usize), + ])) + } + // @block-end sync + + // @block-start run-sync + let run = sync(&client, &latest_chunks).await?; + println!("{run:?}"); + // @block-end run-sync + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts new file mode 100644 index 000000000..dc536dc1c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts @@ -0,0 +1,288 @@ +// @block-start client-connection +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +const QDRANT_URL = process.env.QDRANT_URL; +const QDRANT_API_KEY = process.env.QDRANT_API_KEY; + +const client = new QdrantClient({ + url: QDRANT_URL, + apiKey: QDRANT_API_KEY, +}); +// @block-end client-connection + +// @hide-start +// data and text normalization are not the lesson of this tutorial: +// the full CHUNKS list and normalize() live in the tutorial notebook +const CHUNKS = [ + { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + anchor: "prerequisites", + chunk_num: 0, + text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + }, + { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + anchor: "step-3-enable-an-admin-api-key", + chunk_num: 0, + text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + }, +]; + +function normalize(text: string): string { + return text + .normalize("NFKC") + .replace(/[\u200B\u200C\u200D\uFEFF\u00AD]/g, "") + .replace(/\s+/g, " ") + .trim(); +} +// @hide-end + +// @block-start create-collection +const MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE = "docs-prep-pipeline-v1"; +const COLLECTION = "docs-sync-tutorial"; + +await client.createCollection(COLLECTION, { + vectors: { + size: 384, // all-MiniLM-L6-v2 output dimension + distance: "Cosine", + }, +}); + +await client.updateCollection(COLLECTION, { + metadata: { embedding_model: MODEL, pipeline_version: PIPELINE }, +}); +// @block-end create-collection + +// @block-start check-gate +async function checkGate() { + // compare this pipeline's constants against what the collection records about itself + const meta = ((await client.getCollection(COLLECTION)).config.metadata ?? + {}) as Record; + + if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) { + throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`); + } +} +// @block-end check-gate + +// @block-start identity-and-fingerprint +import { createHash } from "node:crypto"; + +type RawChunk = { url: string; anchor: string; chunk_num: number; text: string }; +type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string }; + +function contentHash(text: string): string { + return createHash("sha256").update(text).digest("hex"); +} + +// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name +function pointId(url: string, anchor: string, num: number): string { + // Qdrant accepts any well-formed UUID as a point ID: + // hash the address, format the digest as a UUID, and the same address always yields the same ID + const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex"); + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`; +} + +// Derive both values (and the section address) for every raw chunk. +function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] { + return chunks.map((c) => { + const text = normalize(c.text); + return { + ...c, + text, + section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url, + content_hash: contentHash(text), + point_id: pointId(c.url, c.anchor, c.chunk_num), + }; + }); +} +// @block-end identity-and-fingerprint + +// @block-start payload +function payload(chunk: SyncChunk, lastUpdated?: string) { + return { + url: chunk.url, + anchor: chunk.anchor, + chunk_num: chunk.chunk_num, + section_url: chunk.section_url, + text: chunk.text, + content_hash: chunk.content_hash, + last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"), + }; +} +// @block-end payload + +// @block-start payload-indexes +for (const field of ["content_hash", "url", "section_url"]) { + await client.createPayloadIndex(COLLECTION, { + field_name: field, + field_schema: "keyword", + }); +} +// @block-end payload-indexes + +// @block-start populate +await client.upsert(COLLECTION, { + points: prepareChunksForSync(CHUNKS).map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, +}); +// @block-end populate + +// @block-start search +const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.query(COLLECTION, { + query: { text: QUERY, model: MODEL }, + limit: 3, + with_payload: ["section_url", "text"], +}); +// @block-end search + +// @hide-start +// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook +const LATEST_CHUNKS = prepareChunksForSync(CHUNKS); +// @hide-end + +// @block-start split-by-state +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async function splitByState(latestChunks: SyncChunk[]) { + const incoming = new Map(latestChunks.map((c) => [c.point_id, c])); + + const stored = new Map(); + const points = await client.retrieve(COLLECTION, { + ids: [...incoming.keys()], + with_payload: ["content_hash"], + with_vector: false, + }); + for (const p of points) { + stored.set(String(p.id), p.payload?.content_hash as string); + } + + const unchanged: SyncChunk[] = []; + const contentChanged: SyncChunk[] = []; + const unknownIds: SyncChunk[] = []; + for (const [pid, c] of incoming) { + if (stored.get(pid) === c.content_hash) { + unchanged.push(c); + } else if (stored.has(pid)) { + contentChanged.push(c); + } else { + unknownIds.push(c); + } + } + + return { incoming, unchanged, contentChanged, unknownIds }; +} + +const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS); +// @block-end split-by-state + +// @block-start re-embed-changed +async function reEmbedChanged(contentChanged: SyncChunk[]) { + if (contentChanged.length === 0) { + return; + } + await client.upsert(COLLECTION, { + points: contentChanged.map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, + }); +} +// @block-end re-embed-changed + +// @block-start reuse-or-add +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async function reuseOrAdd(unknownIds: SyncChunk[]) { + let reused = 0; + let added = 0; + + for (const c of unknownIds) { + const sameText = { + must: [ + { + key: "content_hash", + match: { value: c.content_hash }, + }, + ], + }; + const hits = (await client.scroll(COLLECTION, { + filter: sameText, + limit: 1, + with_payload: ["last_updated"], + with_vector: true, + })).points; + + let point: Schemas["PointStruct"]; + if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated + point = { + id: c.point_id, + vector: hits[0].vector as number[], + payload: payload(c, hits[0].payload?.last_updated as string), + }; + reused += 1; + } else { // genuinely new content: embed and insert + point = { + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + }; + added += 1; + } + + await client.upsert(COLLECTION, { points: [point], wait: true }); + } + + return { reused, added }; +} +// @block-end reuse-or-add + +// @block-start delete-gone +// Remove every point the current crawl no longer contains. Returns how many. +async function deleteGone(incoming: Map) { + if (incoming.size === 0) { + throw new Error("Refusing to delete from an empty source snapshot."); + } + + const stale = { must_not: [{ has_id: [...incoming.keys()] }] }; + + const toDelete = (await client.count(COLLECTION, { filter: stale })).count; + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.delete(COLLECTION, { filter: stale, wait: true }); + return toDelete; +} +// @block-end delete-gone + +// @block-start sync +async function sync(latestChunks: RawChunk[]) { + await checkGate(); // refuse to mix embedding models or pipeline versions + + const chunks = prepareChunksForSync(latestChunks); + const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks); + + await reEmbedChanged(contentChanged); + const { reused, added } = await reuseOrAdd(unknownIds); + const deleted = await deleteGone(incoming); + + return { + "unchanged": unchanged.length, + "re-embedded": contentChanged.length, + "reused_embedding": reused, + "added": added, + "deleted": deleted, + }; +} +// @block-end sync + +// @block-start run-sync +const run = await sync(LATEST_CHUNKS); +console.log(run); +// @block-end run-sync diff --git a/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md b/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md index a91d95998..1189bf834 100644 --- a/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md +++ b/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md @@ -35,27 +35,13 @@ The tutorial has an accompanying [notebook](https://github.com/qdrant/examples/b ## Prerequisites -```python -%pip install -q "qdrant-client>=1.18" -``` +Install the [Qdrant client of your choice](/documentation/interfaces/#client-libraries). We use Qdrant Cloud and its [Free Embedding Inference](/documentation/cloud/inference/#free-embedding-models). Create a Free Tier [Qdrant Cloud cluster](https://cloud.qdrant.io/) and set `QDRANT_URL` and `QDRANT_API_KEY` in your environment. -```python -import os -from qdrant_client import QdrantClient, models - -QDRANT_URL = os.getenv("QDRANT_URL") -QDRANT_API_KEY = os.getenv("QDRANT_API_KEY") - -client = QdrantClient( - url=QDRANT_URL, - api_key=QDRANT_API_KEY, - cloud_inference=True -) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="client-connection" >}} ## The Data: Qdrant Documentation @@ -133,31 +119,11 @@ Vectors produced by different embedding models, or by the same model over differ Let's consider a simple guardrail: save which model and which pipeline version produced the data points, in [**collection metadata**](/documentation/manage-data/collections/#collection-metadata), and verify against it. If one of the two changed, we need to trigger full collection re-embedding. -```python -MODEL = "sentence-transformers/all-MiniLM-L6-v2" -PIPELINE = "docs-prep-pipeline-v1" -COLLECTION = "docs-sync-tutorial" - -client.create_collection( - COLLECTION, - vectors_config=models.VectorParams( - size=384, # all-MiniLM-L6-v2 output dimension - distance=models.Distance.COSINE, - ), - metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE}, -) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="create-collection" >}} The gate against mixing embedding generations is then a simple check at the start of every run: -```python -def check_gate(): - # compare this pipeline's constants against what the collection records about itself - meta = client.get_collection(COLLECTION).config.metadata or {} - - if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE: - raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required") -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="check-gate" >}} ## Characteristics of a Document Chunk @@ -174,32 +140,7 @@ Hence every record should get two derived values: - **Content fingerprint**, like SHA-256 of the text. It changes if a single character changes, and never otherwise. Comparing fingerprints answers "*Is it the same content?*" without comparing texts. - **Deterministic ID** for position in documentation. For example, `url + "#" + anchor + "::" + chunk_num` turned into a UUID, one of the two point ID formats Qdrant accepts. Comparing IDs answers "*Is this content still at the same position?*". -```python -import hashlib -import uuid -from datetime import datetime, timezone - -def content_hash(text): - return hashlib.sha256(text.encode()).hexdigest() - -def point_id(url, anchor, num): - # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name - return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}")) - -def prepare_chunks_for_sync(chunks): - """Derive both values (and the section address) for every raw chunk.""" - out = [] - for c in chunks: - text = normalize(c["text"]) - out.append({ - **c, - "text": text, - "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"], - "content_hash": content_hash(text), - "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]), - }) - return out -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="identity-and-fingerprint" >}} Example: ```text @@ -218,56 +159,24 @@ Additionally, a point can be described by the following fields:
payload() implementation -```python -def payload(chunk, last_updated=None): - return { - "url": chunk["url"], - "anchor": chunk["anchor"], - "chunk_num": chunk["chunk_num"], - "section_url": chunk["section_url"], - "text": chunk["text"], - "content_hash": chunk["content_hash"], - "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"), - } -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload" >}}
For all the payload fields used for filtering or grouping we need to create a [**payload index**](/documentation/manage-data/indexing/). -```python -for field in ("content_hash", "url", "section_url"): - client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload-indexes" >}} ## Populate Collection Populate the collection with the whole documentation. -```python -client.upsert(COLLECTION, points=[ - models.PointStruct( - id=c["point_id"], - vector=models.Document(text=c["text"], model=MODEL), # Cloud Inference embeds text server-side - payload=payload(c), - ) - for c in prepare_chunks_for_sync(CHUNKS) -], wait=True) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="populate" >}}
Test the search against it -```python -QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" - -client.query_points( - COLLECTION, - query=models.Document(text=QUERY, model=MODEL), - limit=3, - with_payload=["section_url", "text"], -) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="search" >}} You should get something like: @@ -354,35 +263,7 @@ We now check every incoming chunk against the collection: does its ID (address) [`retrieve`](/documentation/manage-data/points/) fetches points by ID. At corpus scale you would batch the IDs. -```python -def split_by_state(latest_chunks): - """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.""" - incoming = {c["point_id"]: c for c in latest_chunks} - - stored = {} - points = client.retrieve( - COLLECTION, - ids=list(incoming), - with_payload=["content_hash"], - with_vectors=False, - ) - for p in points: - stored[str(p.id)] = p.payload["content_hash"] - - unchanged, content_changed, unknown_ids = [], [], [] - for pid, c in incoming.items(): - if stored.get(pid) == c["content_hash"]: - unchanged.append(c) - elif pid in stored: - content_changed.append(c) - else: - unknown_ids.append(c) - - return incoming, unchanged, content_changed, unknown_ids - - -incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="split-by-state" >}} ### Case 1: Unchanged, Do Nothing @@ -393,20 +274,7 @@ These chunks carry the same fingerprint as before. The chunk about Step 3 exists under a known ID (it didn't change its position on the docs website) but carries new information. Use `upsert`: writing a point under an existing ID replaces it. -```python -def re_embed_changed(content_changed): - if not content_changed: - return - client.upsert(COLLECTION, - points=[ - models.PointStruct( - id=c["point_id"], - vector=models.Document(text=c["text"], model=MODEL), - payload=payload(c), - ) - for c in content_changed], - wait=True) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="re-embed-changed" >}} ### Cases 3 and 4: ID Is Not Present in the Collection @@ -418,45 +286,7 @@ A filtered [`scroll`](/documentation/manage-data/points/) on `content_hash` answ **Note:** *This version performs one hash lookup per unknown chunk so the decision is easy to inspect. In production, batch hash lookups and point upserts.* -```python -def reuse_or_add(unknown_ids): - """Reuse an existing embedding when the same text is already stored; embed only what is new.""" - reused, added = 0, 0 - - for c in unknown_ids: - same_text = models.Filter(must=[ - models.FieldCondition( - key="content_hash", - match=models.MatchValue(value=c["content_hash"]), - ) - ]) - hits, _ = client.scroll( - COLLECTION, - scroll_filter=same_text, - limit=1, - with_payload=["last_updated"], - with_vectors=True, - ) - - if hits: # same text, new address: copy the vector, keep its last_updated - point = models.PointStruct( - id=c["point_id"], - vector=hits[0].vector, - payload=payload(c, hits[0].payload["last_updated"]), - ) - reused += 1 - else: # genuinely new content: embed and insert - point = models.PointStruct( - id=c["point_id"], - vector=models.Document(text=c["text"], model=MODEL), - payload=payload(c), - ) - added += 1 - - client.upsert(COLLECTION, points=[point], wait=True) - - return reused, added -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="reuse-or-add" >}} What's important to notice: the old points, the migration page under its old URL, are still in the collection. They need to be removed, and that is the last case. @@ -471,51 +301,17 @@ Whatever LATEST_CHUNKS does not contain no longer exists at the source. The dele **Note:** Frequent re-embeddings and deletions don't degrade the index over time: background [optimizers](/documentation/ops-optimization/optimizer/) rebuild and merge index segments as changes accumulate. -```python -def delete_gone(incoming_ids): - """Remove every point the current crawl no longer contains. Returns how many.""" - if not incoming_ids: - raise ValueError("Refusing to delete from an empty source snapshot.") - - stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))]) - - to_delete = client.count(COLLECTION, count_filter=stale).count - - # potential check against a threshold to avoid accidental mass deletion could be added here - client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True) - return to_delete -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="delete-gone" >}} ## Run and Verify the Sync The five cases, assembled from the functions defined above: -```python -def sync(latest_chunks): - check_gate() # refuse to mix embedding models or pipeline versions - - chunks = prepare_chunks_for_sync(latest_chunks) - incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks) - - re_embed_changed(content_changed) - reused, added = reuse_or_add(unknown_ids) - deleted = delete_gone(incoming_ids) - - return { - "unchanged": len(unchanged), - "re-embedded": len(content_changed), - "reused_embedding": reused, - "added": added, - "deleted": deleted, - } -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="sync" >}} Run the sync. -```python -run = sync(LATEST_CHUNKS) -print(run) -``` +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="run-sync" >}} You should see something like: