added snippets in all languages

This commit is contained in:
Evgeniya Sukhodolskaya
2026-07-21 18:55:24 +02:00
parent 2d2cd2d1f7
commit 169c167102
99 changed files with 5268 additions and 221 deletions
@@ -15,4 +15,5 @@ serde_json = "1.0.145"
tempfile = "3"
tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] }
ureq = { version = "3", features = ["json"] }
uuid = { version = "1.18.1", features = ["v4"] }
uuid = { version = "1.18.1", features = ["v4", "v5"] }
sha2 = "0.11"
@@ -6,6 +6,6 @@
| [Time-Based Sharding](/documentation/tutorials-operations/time-based-sharding/) | Efficiently manage time-series data with user-defined sharding. | <span class="pill">Any</span> | 1h | <span class="text-yellow">Intermediate</span> |
| [Large-Scale Search](/documentation/tutorials-operations/large-scale-search/) | Cost-efficient search for LAION-400M datasets. | <span class="pill">Any</span> | 48h | <span class="text-red">Advanced</span> |
| [Secure a Self-Hosted Instance](/documentation/tutorials-operations/secure-qdrant/) | Enable TLS, API keys, and JWT access control. | <span class="pill">Any</span> | 45m | <span class="text-yellow">Intermediate</span> |
| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | <span class="pill">Python</span> | 25m | <span class="text-green">Beginner</span> |
| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | <span class="pill">Any</span> | 25m | <span class="text-green">Beginner</span> |
| [Qdrant Cloud Prometheus Monitoring](/documentation/ops-monitoring/managed-cloud-prometheus/) | Observability with Prometheus and Grafana. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
| [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
@@ -0,0 +1,310 @@
using System.Security.Cryptography;
using System.Text;
using System.Text.RegularExpressions;
using Qdrant.Client;
using Qdrant.Client.Grpc;
using static Qdrant.Client.Grpc.Conditions;
using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
public class Snippet
{
public static async Task Run()
{
// @block-start client-connection
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
var client = new QdrantClient(
host: QDRANT_URL!,
https: true,
apiKey: QDRANT_API_KEY
);
// @block-end client-connection
// @hide-start
// data and text normalization are not the lesson of this tutorial:
// the full CHUNKS list and Normalize() live in the tutorial notebook
var CHUNKS = new List<Chunk>
{
(
Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
Anchor: "prerequisites",
ChunkNum: 0,
Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
SectionUrl: "", ContentHash: "", PointId: ""
),
(
Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
Anchor: "step-3-enable-an-admin-api-key",
ChunkNum: 0,
Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
SectionUrl: "", ContentHash: "", PointId: ""
),
};
string Normalize(string text) => Regex.Replace(text, @"\s+", " ").Trim();
// @hide-end
// @block-start create-collection
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
var PIPELINE = "docs-prep-pipeline-v1";
var COLLECTION = "docs-sync-tutorial";
await client.CreateCollectionAsync(
collectionName: COLLECTION,
vectorsConfig: new VectorParams
{
Size = 384, // all-MiniLM-L6-v2 output dimension
Distance = Distance.Cosine
},
metadata: new()
{
["embedding_model"] = MODEL,
["pipeline_version"] = PIPELINE
}
);
// @block-end create-collection
// @block-start check-gate
async Task CheckGate()
{
// compare this pipeline's constants against what the collection records about itself
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
if (model != MODEL || pipeline != PIPELINE)
throw new InvalidOperationException(
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
}
// @block-end check-gate
// @block-start identity-and-fingerprint
string ContentHash(string text) =>
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
// Qdrant accepts any well-formed UUID as a point ID:
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
string PointIdFor(string url, string anchor, int num) =>
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
// Derive both values (and the section address) for every raw chunk.
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
{
var prepared = new List<Chunk>();
foreach (var c in chunks)
{
var text = Normalize(c.Text);
prepared.Add(c with
{
Text = text,
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
ContentHash = ContentHash(text),
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
});
}
return prepared;
}
// @block-end identity-and-fingerprint
// @block-start payload
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
{
["url"] = chunk.Url,
["anchor"] = chunk.Anchor,
["chunk_num"] = chunk.ChunkNum,
["section_url"] = chunk.SectionUrl,
["text"] = chunk.Text,
["content_hash"] = chunk.ContentHash,
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
};
// @block-end payload
// @block-start payload-indexes
foreach (var field in new[] { "content_hash", "url", "section_url" })
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
// @block-end payload-indexes
// @block-start populate
await client.UpsertAsync(
collectionName: COLLECTION,
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
}).ToList(),
wait: true
);
// @block-end populate
// @block-start search
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
await client.QueryAsync(
collectionName: COLLECTION,
query: new Document { Text = QUERY, Model = MODEL },
limit: 3,
payloadSelector: new[] { "section_url", "text" }
);
// @block-end search
// @hide-start
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
var LATEST_CHUNKS = PrepareChunksForSync(CHUNKS);
// @hide-end
// @block-start split-by-state
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
SplitByState(List<Chunk> latestChunks)
{
var incoming = latestChunks.ToDictionary(c => c.PointId);
var stored = new Dictionary<string, string>();
var points = await client.RetrieveAsync(
COLLECTION,
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
payloadSelector: new[] { "content_hash" },
vectorSelector: false
);
foreach (var p in points)
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
var unchanged = new List<Chunk>();
var contentChanged = new List<Chunk>();
var unknownIds = new List<Chunk>();
foreach (var (pid, c) in incoming)
{
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
unchanged.Add(c);
else if (stored.ContainsKey(pid))
contentChanged.Add(c);
else
unknownIds.Add(c);
}
return (incoming, unchanged, contentChanged, unknownIds);
}
var splitState = await SplitByState(LATEST_CHUNKS);
// @block-end split-by-state
// @block-start re-embed-changed
async Task ReEmbedChanged(List<Chunk> contentChanged)
{
if (contentChanged.Count == 0)
return;
await client.UpsertAsync(
collectionName: COLLECTION,
points: contentChanged.Select(c => new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
}).ToList(),
wait: true
);
}
// @block-end re-embed-changed
// @block-start reuse-or-add
// Reuse an existing embedding when the same text is already stored; embed only what is new.
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
{
int reused = 0, added = 0;
foreach (var c in unknownIds)
{
var sameText = new Filter
{
Must = { MatchKeyword("content_hash", c.ContentHash) }
};
var hits = (await client.ScrollAsync(
COLLECTION,
filter: sameText,
limit: 1,
payloadSelector: new[] { "last_updated" },
vectorsSelector: true
)).Result;
PointStruct point;
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
{
point = new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
};
reused++;
}
else // genuinely new content: embed and insert
{
point = new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
};
added++;
}
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
}
return (reused, added);
}
// @block-end reuse-or-add
// @block-start delete-gone
// Remove every point the current crawl no longer contains. Returns how many.
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
{
if (incomingIds.Count == 0)
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
var stale = new Filter
{
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
};
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
// potential check against a threshold to avoid accidental mass deletion could be added here
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
return toDelete;
}
// @block-end delete-gone
// @block-start sync
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
{
await CheckGate(); // refuse to mix embedding models or pipeline versions
var chunks = PrepareChunksForSync(latestChunks);
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
await ReEmbedChanged(contentChanged);
var (reused, added) = await ReuseOrAdd(unknownIds);
var deleted = await DeleteGone(incomingIds);
return new Dictionary<string, long>
{
["unchanged"] = unchanged.Count,
["re-embedded"] = contentChanged.Count,
["reused_embedding"] = reused,
["added"] = added,
["deleted"] = (long)deleted,
};
}
// @block-end sync
// @block-start run-sync
var run = await Sync(LATEST_CHUNKS);
foreach (var (op, count) in run)
Console.WriteLine($"{op}: {count}");
// @block-end run-sync
}
}
@@ -0,0 +1,13 @@
```csharp
async Task CheckGate()
{
// compare this pipeline's constants against what the collection records about itself
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
if (model != MODEL || pipeline != PIPELINE)
throw new InvalidOperationException(
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
}
```
@@ -0,0 +1,11 @@
```go
checkGate := func() {
// compare this pipeline's constants against what the collection records about itself
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
meta := info.GetConfig().GetMetadata()
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
}
}
```
@@ -0,0 +1,15 @@
```java
static void checkGate() throws Exception {
// compare this pipeline's constants against what the collection records about itself
Map<String, Value> meta =
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
Value model = meta.get("embedding_model");
Value pipeline = meta.get("pipeline_version");
if (model == null || !MODEL.equals(model.getStringValue())
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
throw new RuntimeException(
"collection was built by " + meta + ": full re-embed into a fresh collection required");
}
}
```
@@ -0,0 +1,8 @@
```python
def check_gate():
# compare this pipeline's constants against what the collection records about itself
meta = client.get_collection(COLLECTION).config.metadata or {}
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
```
@@ -0,0 +1,22 @@
```rust
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
// compare this pipeline's constants against what the collection records about itself
let meta = client
.collection_info(COLLECTION)
.await?
.result
.and_then(|info| info.config)
.map(|config| config.metadata)
.unwrap_or_default();
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
!= Some(PIPELINE)
{
anyhow::bail!(
"collection was built by {meta:?}: full re-embed into a fresh collection required"
);
}
Ok(())
}
```
@@ -0,0 +1,11 @@
```typescript
async function checkGate() {
// compare this pipeline's constants against what the collection records about itself
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
{}) as Record<string, unknown>;
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
}
}
```
@@ -0,0 +1,10 @@
```csharp
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
var client = new QdrantClient(
host: QDRANT_URL!,
https: true,
apiKey: QDRANT_API_KEY
);
```
@@ -0,0 +1,10 @@
```go
QDRANT_URL := os.Getenv("QDRANT_URL")
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
client, err := qdrant.NewClient(&qdrant.Config{
Host: QDRANT_URL,
APIKey: QDRANT_API_KEY,
UseTLS: true,
})
```
@@ -0,0 +1,10 @@
```java
static final String QDRANT_URL = System.getenv("QDRANT_URL");
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
static final QdrantClient client =
new QdrantClient(
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
.withApiKey(QDRANT_API_KEY)
.build());
```
@@ -0,0 +1,14 @@
```python
import os
from qdrant_client import QdrantClient, models
QDRANT_URL = os.getenv("QDRANT_URL")
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
client = QdrantClient(
url=QDRANT_URL,
api_key=QDRANT_API_KEY,
cloud_inference=True
)
```
@@ -0,0 +1,8 @@
```rust
let qdrant_url = std::env::var("QDRANT_URL")?;
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
let client = Qdrant::from_url(&qdrant_url)
.api_key(qdrant_api_key)
.build()?;
```
@@ -0,0 +1,11 @@
```typescript
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
const QDRANT_URL = process.env.QDRANT_URL;
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
const client = new QdrantClient({
url: QDRANT_URL,
apiKey: QDRANT_API_KEY,
});
```
@@ -0,0 +1,19 @@
```csharp
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
var PIPELINE = "docs-prep-pipeline-v1";
var COLLECTION = "docs-sync-tutorial";
await client.CreateCollectionAsync(
collectionName: COLLECTION,
vectorsConfig: new VectorParams
{
Size = 384, // all-MiniLM-L6-v2 output dimension
Distance = Distance.Cosine
},
metadata: new()
{
["embedding_model"] = MODEL,
["pipeline_version"] = PIPELINE
}
);
```
@@ -0,0 +1,17 @@
```go
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
PIPELINE := "docs-prep-pipeline-v1"
COLLECTION := "docs-sync-tutorial"
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
CollectionName: COLLECTION,
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
Size: 384, // all-MiniLM-L6-v2 output dimension
Distance: qdrant.Distance_Cosine,
}),
Metadata: qdrant.NewValueMap(map[string]any{
"embedding_model": MODEL,
"pipeline_version": PIPELINE,
}),
})
```
@@ -0,0 +1,24 @@
```java
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
static final String PIPELINE = "docs-prep-pipeline-v1";
static final String COLLECTION = "docs-sync-tutorial";
static void createCollection() throws Exception {
client.createCollectionAsync(
CreateCollection.newBuilder()
.setCollectionName(COLLECTION)
.setVectorsConfig(
VectorsConfig.newBuilder()
.setParams(
VectorParams.newBuilder()
.setSize(384) // all-MiniLM-L6-v2 output dimension
.setDistance(Distance.Cosine)
.build())
.build())
.putAllMetadata(
Map.of(
"embedding_model", value(MODEL),
"pipeline_version", value(PIPELINE)))
.build()).get();
}
```
@@ -0,0 +1,14 @@
```python
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
PIPELINE = "docs-prep-pipeline-v1"
COLLECTION = "docs-sync-tutorial"
client.create_collection(
COLLECTION,
vectors_config=models.VectorParams(
size=384, # all-MiniLM-L6-v2 output dimension
distance=models.Distance.COSINE,
),
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
)
```
@@ -0,0 +1,20 @@
```rust
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
const PIPELINE: &str = "docs-prep-pipeline-v1";
const COLLECTION: &str = "docs-sync-tutorial";
let mut metadata: HashMap<String, Value> = HashMap::new();
metadata.insert("embedding_model".to_string(), json!(MODEL));
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
client
.create_collection(
CreateCollectionBuilder::new(COLLECTION)
.vectors_config(VectorParamsBuilder::new(
384, // all-MiniLM-L6-v2 output dimension
Distance::Cosine,
))
.metadata(metadata),
)
.await?;
```
@@ -0,0 +1,16 @@
```typescript
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
const PIPELINE = "docs-prep-pipeline-v1";
const COLLECTION = "docs-sync-tutorial";
await client.createCollection(COLLECTION, {
vectors: {
size: 384, // all-MiniLM-L6-v2 output dimension
distance: "Cosine",
},
});
await client.updateCollection(COLLECTION, {
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
});
```
@@ -0,0 +1,248 @@
```csharp
using System.Security.Cryptography;
using System.Text;
using System.Text.RegularExpressions;
using Qdrant.Client;
using Qdrant.Client.Grpc;
using static Qdrant.Client.Grpc.Conditions;
using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
var client = new QdrantClient(
host: QDRANT_URL!,
https: true,
apiKey: QDRANT_API_KEY
);
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
var PIPELINE = "docs-prep-pipeline-v1";
var COLLECTION = "docs-sync-tutorial";
await client.CreateCollectionAsync(
collectionName: COLLECTION,
vectorsConfig: new VectorParams
{
Size = 384, // all-MiniLM-L6-v2 output dimension
Distance = Distance.Cosine
},
metadata: new()
{
["embedding_model"] = MODEL,
["pipeline_version"] = PIPELINE
}
);
async Task CheckGate()
{
// compare this pipeline's constants against what the collection records about itself
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
if (model != MODEL || pipeline != PIPELINE)
throw new InvalidOperationException(
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
}
string ContentHash(string text) =>
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
// Qdrant accepts any well-formed UUID as a point ID:
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
string PointIdFor(string url, string anchor, int num) =>
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
// Derive both values (and the section address) for every raw chunk.
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
{
var prepared = new List<Chunk>();
foreach (var c in chunks)
{
var text = Normalize(c.Text);
prepared.Add(c with
{
Text = text,
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
ContentHash = ContentHash(text),
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
});
}
return prepared;
}
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
{
["url"] = chunk.Url,
["anchor"] = chunk.Anchor,
["chunk_num"] = chunk.ChunkNum,
["section_url"] = chunk.SectionUrl,
["text"] = chunk.Text,
["content_hash"] = chunk.ContentHash,
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
};
foreach (var field in new[] { "content_hash", "url", "section_url" })
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
await client.UpsertAsync(
collectionName: COLLECTION,
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
}).ToList(),
wait: true
);
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
await client.QueryAsync(
collectionName: COLLECTION,
query: new Document { Text = QUERY, Model = MODEL },
limit: 3,
payloadSelector: new[] { "section_url", "text" }
);
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
SplitByState(List<Chunk> latestChunks)
{
var incoming = latestChunks.ToDictionary(c => c.PointId);
var stored = new Dictionary<string, string>();
var points = await client.RetrieveAsync(
COLLECTION,
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
payloadSelector: new[] { "content_hash" },
vectorSelector: false
);
foreach (var p in points)
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
var unchanged = new List<Chunk>();
var contentChanged = new List<Chunk>();
var unknownIds = new List<Chunk>();
foreach (var (pid, c) in incoming)
{
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
unchanged.Add(c);
else if (stored.ContainsKey(pid))
contentChanged.Add(c);
else
unknownIds.Add(c);
}
return (incoming, unchanged, contentChanged, unknownIds);
}
var splitState = await SplitByState(LATEST_CHUNKS);
async Task ReEmbedChanged(List<Chunk> contentChanged)
{
if (contentChanged.Count == 0)
return;
await client.UpsertAsync(
collectionName: COLLECTION,
points: contentChanged.Select(c => new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
}).ToList(),
wait: true
);
}
// Reuse an existing embedding when the same text is already stored; embed only what is new.
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
{
int reused = 0, added = 0;
foreach (var c in unknownIds)
{
var sameText = new Filter
{
Must = { MatchKeyword("content_hash", c.ContentHash) }
};
var hits = (await client.ScrollAsync(
COLLECTION,
filter: sameText,
limit: 1,
payloadSelector: new[] { "last_updated" },
vectorsSelector: true
)).Result;
PointStruct point;
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
{
point = new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
};
reused++;
}
else // genuinely new content: embed and insert
{
point = new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
};
added++;
}
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
}
return (reused, added);
}
// Remove every point the current crawl no longer contains. Returns how many.
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
{
if (incomingIds.Count == 0)
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
var stale = new Filter
{
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
};
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
// potential check against a threshold to avoid accidental mass deletion could be added here
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
return toDelete;
}
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
{
await CheckGate(); // refuse to mix embedding models or pipeline versions
var chunks = PrepareChunksForSync(latestChunks);
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
await ReEmbedChanged(contentChanged);
var (reused, added) = await ReuseOrAdd(unknownIds);
var deleted = await DeleteGone(incomingIds);
return new Dictionary<string, long>
{
["unchanged"] = unchanged.Count,
["re-embedded"] = contentChanged.Count,
["reused_embedding"] = reused,
["added"] = added,
["deleted"] = (long)deleted,
};
}
var run = await Sync(LATEST_CHUNKS);
foreach (var (op, count) in run)
Console.WriteLine($"{op}: {count}");
```
@@ -0,0 +1,19 @@
```csharp
// Remove every point the current crawl no longer contains. Returns how many.
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
{
if (incomingIds.Count == 0)
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
var stale = new Filter
{
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
};
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
// potential check against a threshold to avoid accidental mass deletion could be added here
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
return toDelete;
}
```
@@ -0,0 +1,29 @@
```go
// remove every point the current crawl no longer contains, return how many
deleteGone := func(incomingIDs map[string]Chunk) int {
if len(incomingIDs) == 0 {
panic("Refusing to delete from an empty source snapshot.")
}
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
for pid := range incomingIDs {
ids = append(ids, qdrant.NewID(pid))
}
stale := &qdrant.Filter{
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
}
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
CollectionName: COLLECTION,
Filter: stale,
})
// potential check against a threshold to avoid accidental mass deletion could be added here
client.Delete(context.Background(), &qdrant.DeletePoints{
CollectionName: COLLECTION,
Points: qdrant.NewPointsSelectorFilter(stale),
Wait: qdrant.PtrOf(true),
})
return int(toDelete)
}
```
@@ -0,0 +1,21 @@
```java
// Remove every point the current crawl no longer contains. Returns how many.
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
if (incomingIds.isEmpty()) {
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
}
Filter stale = Filter.newBuilder()
.addMustNot(hasId(
incomingIds.keySet().stream()
.map(pid -> id(UUID.fromString(pid)))
.collect(Collectors.toList())))
.build();
long toDelete = client.countAsync(COLLECTION, stale, true).get();
// potential check against a threshold to avoid accidental mass deletion could be added here
client.deleteAsync(COLLECTION, stale).get();
return toDelete;
}
```
@@ -0,0 +1,14 @@
```python
def delete_gone(incoming_ids):
"""Remove every point the current crawl no longer contains. Returns how many."""
if not incoming_ids:
raise ValueError("Refusing to delete from an empty source snapshot.")
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
to_delete = client.count(COLLECTION, count_filter=stale).count
# potential check against a threshold to avoid accidental mass deletion could be added here
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
return to_delete
```
@@ -0,0 +1,28 @@
```rust
/// Remove every point the current crawl no longer contains. Returns how many.
async fn delete_gone(
client: &Qdrant,
incoming_ids: &HashMap<String, Chunk>,
) -> anyhow::Result<u64> {
if incoming_ids.is_empty() {
anyhow::bail!("Refusing to delete from an empty source snapshot.");
}
let stale = Filter::must_not([Condition::has_id(
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
)]);
let to_delete = client
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
.await?
.result
.map(|r| r.count)
.unwrap_or(0);
// potential check against a threshold to avoid accidental mass deletion could be added here
client
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
.await?;
Ok(to_delete)
}
```
@@ -0,0 +1,16 @@
```typescript
// Remove every point the current crawl no longer contains. Returns how many.
async function deleteGone(incoming: Map<string, SyncChunk>) {
if (incoming.size === 0) {
throw new Error("Refusing to delete from an empty source snapshot.");
}
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
// potential check against a threshold to avoid accidental mass deletion could be added here
await client.delete(COLLECTION, { filter: stale, wait: true });
return toDelete;
}
```
@@ -0,0 +1,276 @@
```go
import (
"context"
"crypto/sha256"
"encoding/hex"
"fmt"
"os"
"regexp"
"strings"
"time"
"github.com/google/uuid"
"github.com/qdrant/go-client/qdrant"
)
QDRANT_URL := os.Getenv("QDRANT_URL")
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
client, err := qdrant.NewClient(&qdrant.Config{
Host: QDRANT_URL,
APIKey: QDRANT_API_KEY,
UseTLS: true,
})
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
PIPELINE := "docs-prep-pipeline-v1"
COLLECTION := "docs-sync-tutorial"
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
CollectionName: COLLECTION,
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
Size: 384, // all-MiniLM-L6-v2 output dimension
Distance: qdrant.Distance_Cosine,
}),
Metadata: qdrant.NewValueMap(map[string]any{
"embedding_model": MODEL,
"pipeline_version": PIPELINE,
}),
})
checkGate := func() {
// compare this pipeline's constants against what the collection records about itself
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
meta := info.GetConfig().GetMetadata()
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
}
}
contentHash := func(text string) string {
sum := sha256.Sum256([]byte(text))
return hex.EncodeToString(sum[:])
}
pointID := func(url, anchor string, num int) string {
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
// marking the input as a URL-like name
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
}
// derive both values (and the section address) for every raw chunk
prepareChunksForSync := func(chunks []Chunk) []Chunk {
out := make([]Chunk, 0, len(chunks))
for _, c := range chunks {
c.Text = normalize(c.Text)
c.SectionURL = c.URL
if c.Anchor != "" {
c.SectionURL = c.URL + "#" + c.Anchor
}
c.ContentHash = contentHash(c.Text)
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
out = append(out, c)
}
return out
}
payload := func(c Chunk, lastUpdated string) map[string]any {
if lastUpdated == "" {
lastUpdated = time.Now().UTC().Format(time.RFC3339)
}
return map[string]any{
"url": c.URL,
"anchor": c.Anchor,
"chunk_num": c.ChunkNum,
"section_url": c.SectionURL,
"text": c.Text,
"content_hash": c.ContentHash,
"last_updated": lastUpdated,
}
}
for _, field := range []string{"content_hash", "url", "section_url"} {
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
CollectionName: COLLECTION,
FieldName: field,
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
})
}
var points []*qdrant.PointStruct
for _, c := range prepareChunksForSync(CHUNKS) {
points = append(points, &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
})
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: points,
Wait: qdrant.PtrOf(true),
})
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: COLLECTION,
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
Limit: qdrant.PtrOf(uint64(3)),
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
})
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
incoming := make(map[string]Chunk, len(latestChunks))
ids := make([]*qdrant.PointId, 0, len(latestChunks))
for _, c := range latestChunks {
incoming[c.PointID] = c
ids = append(ids, qdrant.NewID(c.PointID))
}
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
CollectionName: COLLECTION,
Ids: ids,
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
WithVectors: qdrant.NewWithVectors(false),
})
stored := make(map[string]string, len(retrieved))
for _, p := range retrieved {
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
}
var unchanged, contentChanged, unknownIDs []Chunk
for pid, c := range incoming {
storedHash, found := stored[pid]
switch {
case found && storedHash == c.ContentHash:
unchanged = append(unchanged, c)
case found:
contentChanged = append(contentChanged, c)
default:
unknownIDs = append(unknownIDs, c)
}
}
return incoming, unchanged, contentChanged, unknownIDs
}
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
reEmbedChanged := func(contentChanged []Chunk) {
if len(contentChanged) == 0 {
return
}
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
for _, c := range contentChanged {
points = append(points, &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
})
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: points,
Wait: qdrant.PtrOf(true),
})
}
// reuse an existing embedding when the same text is already stored; embed only what is new
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
reused, added := 0, 0
for _, c := range unknownIDs {
sameText := &qdrant.Filter{
Must: []*qdrant.Condition{
qdrant.NewMatch("content_hash", c.ContentHash),
},
}
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
CollectionName: COLLECTION,
Filter: sameText,
Limit: qdrant.PtrOf(uint32(1)),
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
WithVectors: qdrant.NewWithVectors(true),
})
var point *qdrant.PointStruct
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
point = &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
}
reused++
} else { // genuinely new content: embed and insert
point = &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
}
added++
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: []*qdrant.PointStruct{point},
Wait: qdrant.PtrOf(true),
})
}
return reused, added
}
// remove every point the current crawl no longer contains, return how many
deleteGone := func(incomingIDs map[string]Chunk) int {
if len(incomingIDs) == 0 {
panic("Refusing to delete from an empty source snapshot.")
}
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
for pid := range incomingIDs {
ids = append(ids, qdrant.NewID(pid))
}
stale := &qdrant.Filter{
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
}
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
CollectionName: COLLECTION,
Filter: stale,
})
// potential check against a threshold to avoid accidental mass deletion could be added here
client.Delete(context.Background(), &qdrant.DeletePoints{
CollectionName: COLLECTION,
Points: qdrant.NewPointsSelectorFilter(stale),
Wait: qdrant.PtrOf(true),
})
return int(toDelete)
}
sync := func(latestChunks []Chunk) map[string]int {
checkGate() // refuse to mix embedding models or pipeline versions
chunks := prepareChunksForSync(latestChunks)
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
reEmbedChanged(contentChanged)
reused, added := reuseOrAdd(unknownIDs)
deleted := deleteGone(incomingIDs)
return map[string]int{
"unchanged": len(unchanged),
"re-embedded": len(contentChanged),
"reused_embedding": reused,
"added": added,
"deleted": deleted,
}
}
run := sync(LATEST_CHUNKS)
fmt.Println(run)
```
@@ -0,0 +1,27 @@
```csharp
string ContentHash(string text) =>
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
// Qdrant accepts any well-formed UUID as a point ID:
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
string PointIdFor(string url, string anchor, int num) =>
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
// Derive both values (and the section address) for every raw chunk.
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
{
var prepared = new List<Chunk>();
foreach (var c in chunks)
{
var text = Normalize(c.Text);
prepared.Add(c with
{
Text = text,
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
ContentHash = ContentHash(text),
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
});
}
return prepared;
}
```
@@ -0,0 +1,28 @@
```go
contentHash := func(text string) string {
sum := sha256.Sum256([]byte(text))
return hex.EncodeToString(sum[:])
}
pointID := func(url, anchor string, num int) string {
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
// marking the input as a URL-like name
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
}
// derive both values (and the section address) for every raw chunk
prepareChunksForSync := func(chunks []Chunk) []Chunk {
out := make([]Chunk, 0, len(chunks))
for _, c := range chunks {
c.Text = normalize(c.Text)
c.SectionURL = c.URL
if c.Anchor != "" {
c.SectionURL = c.URL + "#" + c.Anchor
}
c.ContentHash = contentHash(c.Text)
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
out = append(out, c)
}
return out
}
```
@@ -0,0 +1,27 @@
```java
static String contentHash(String text) throws Exception {
byte[] digest = MessageDigest.getInstance("SHA-256")
.digest(text.getBytes(StandardCharsets.UTF_8));
return String.format("%064x", new BigInteger(1, digest));
}
static String pointId(String url, String anchor, int num) {
// name-based UUID (version 3); the same address always yields the same ID
return UUID.nameUUIDFromBytes(
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
}
// Derive both values (and the section address) for every raw chunk.
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
List<Chunk> out = new ArrayList<>();
for (Chunk c : chunks) {
String text = normalize(c.text);
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
prepared.contentHash = contentHash(text);
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
out.add(prepared);
}
return out;
}
```
@@ -0,0 +1,26 @@
```python
import hashlib
import uuid
from datetime import datetime, timezone
def content_hash(text):
return hashlib.sha256(text.encode()).hexdigest()
def point_id(url, anchor, num):
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
def prepare_chunks_for_sync(chunks):
"""Derive both values (and the section address) for every raw chunk."""
out = []
for c in chunks:
text = normalize(c["text"])
out.append({
**c,
"text": text,
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
"content_hash": content_hash(text),
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
})
return out
```
@@ -0,0 +1,38 @@
```rust
fn content_hash(text: &str) -> String {
Sha256::digest(text.as_bytes())
.iter()
.map(|byte| format!("{byte:02x}"))
.collect()
}
fn point_id(url: &str, anchor: &str, num: u32) -> String {
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
uuid::Uuid::new_v5(
&uuid::Uuid::NAMESPACE_URL,
format!("{url}#{anchor}::{num}").as_bytes(),
)
.to_string()
}
/// Derive both values (and the section address) for every raw chunk.
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
chunks
.iter()
.map(|c| {
let text = normalize(&c.text);
Chunk {
text: text.clone(),
section_url: if c.anchor.is_empty() {
c.url.clone()
} else {
format!("{}#{}", c.url, c.anchor)
},
content_hash: content_hash(&text),
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
..c.clone()
}
})
.collect()
}
```
@@ -0,0 +1,32 @@
```typescript
import { createHash } from "node:crypto";
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
function contentHash(text: string): string {
return createHash("sha256").update(text).digest("hex");
}
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
function pointId(url: string, anchor: string, num: number): string {
// Qdrant accepts any well-formed UUID as a point ID:
// hash the address, format the digest as a UUID, and the same address always yields the same ID
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
}
// Derive both values (and the section address) for every raw chunk.
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
return chunks.map((c) => {
const text = normalize(c.text);
return {
...c,
text,
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
content_hash: contentHash(text),
point_id: pointId(c.url, c.anchor, c.chunk_num),
};
});
}
```
@@ -0,0 +1,328 @@
```java
import static io.qdrant.client.ConditionFactory.hasId;
import static io.qdrant.client.ConditionFactory.matchKeyword;
import static io.qdrant.client.PointIdFactory.id;
import static io.qdrant.client.QueryFactory.nearest;
import static io.qdrant.client.ValueFactory.value;
import static io.qdrant.client.VectorFactory.vector;
import static io.qdrant.client.VectorsFactory.vectors;
import io.qdrant.client.QdrantClient;
import io.qdrant.client.QdrantGrpcClient;
import io.qdrant.client.VectorOutputHelper;
import io.qdrant.client.WithPayloadSelectorFactory;
import io.qdrant.client.WithVectorsSelectorFactory;
import io.qdrant.client.grpc.Collections.CreateCollection;
import io.qdrant.client.grpc.Collections.Distance;
import io.qdrant.client.grpc.Collections.PayloadSchemaType;
import io.qdrant.client.grpc.Collections.VectorParams;
import io.qdrant.client.grpc.Collections.VectorsConfig;
import io.qdrant.client.grpc.Common.Filter;
import io.qdrant.client.grpc.JsonWithInt.Value;
import io.qdrant.client.grpc.Points.Document;
import io.qdrant.client.grpc.Points.PointStruct;
import io.qdrant.client.grpc.Points.QueryPoints;
import io.qdrant.client.grpc.Points.ScrollPoints;
import java.math.BigInteger;
import java.nio.charset.StandardCharsets;
import java.security.MessageDigest;
import java.time.OffsetDateTime;
import java.time.ZoneOffset;
import java.time.temporal.ChronoUnit;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.UUID;
import java.util.stream.Collectors;
static final String QDRANT_URL = System.getenv("QDRANT_URL");
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
static final QdrantClient client =
new QdrantClient(
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
.withApiKey(QDRANT_API_KEY)
.build());
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
static final String PIPELINE = "docs-prep-pipeline-v1";
static final String COLLECTION = "docs-sync-tutorial";
static void createCollection() throws Exception {
client.createCollectionAsync(
CreateCollection.newBuilder()
.setCollectionName(COLLECTION)
.setVectorsConfig(
VectorsConfig.newBuilder()
.setParams(
VectorParams.newBuilder()
.setSize(384) // all-MiniLM-L6-v2 output dimension
.setDistance(Distance.Cosine)
.build())
.build())
.putAllMetadata(
Map.of(
"embedding_model", value(MODEL),
"pipeline_version", value(PIPELINE)))
.build()).get();
}
static void checkGate() throws Exception {
// compare this pipeline's constants against what the collection records about itself
Map<String, Value> meta =
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
Value model = meta.get("embedding_model");
Value pipeline = meta.get("pipeline_version");
if (model == null || !MODEL.equals(model.getStringValue())
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
throw new RuntimeException(
"collection was built by " + meta + ": full re-embed into a fresh collection required");
}
}
static String contentHash(String text) throws Exception {
byte[] digest = MessageDigest.getInstance("SHA-256")
.digest(text.getBytes(StandardCharsets.UTF_8));
return String.format("%064x", new BigInteger(1, digest));
}
static String pointId(String url, String anchor, int num) {
// name-based UUID (version 3); the same address always yields the same ID
return UUID.nameUUIDFromBytes(
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
}
// Derive both values (and the section address) for every raw chunk.
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
List<Chunk> out = new ArrayList<>();
for (Chunk c : chunks) {
String text = normalize(c.text);
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
prepared.contentHash = contentHash(text);
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
out.add(prepared);
}
return out;
}
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
Map<String, Value> p = new HashMap<>();
p.put("url", value(chunk.url));
p.put("anchor", value(chunk.anchor));
p.put("chunk_num", value(chunk.chunkNum));
p.put("section_url", value(chunk.sectionUrl));
p.put("text", value(chunk.text));
p.put("content_hash", value(chunk.contentHash));
p.put("last_updated", value(lastUpdated != null
? lastUpdated
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
return p;
}
static void createPayloadIndexes() throws Exception {
for (String field : List.of("content_hash", "url", "section_url")) {
client.createPayloadIndexAsync(
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
}
}
static void populate() throws Exception {
List<PointStruct> points = new ArrayList<>();
for (Chunk c : prepareChunksForSync(CHUNKS)) {
points.add(
PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build());
}
client.upsertAsync(COLLECTION, points).get();
}
static final String QUERY =
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
static void search() throws Exception {
client.queryAsync(
QueryPoints.newBuilder()
.setCollectionName(COLLECTION)
.setQuery(
nearest(
Document.newBuilder()
.setText(QUERY)
.setModel(MODEL)
.build()))
.setLimit(3)
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
.build()).get();
}
static class SyncState {
Map<String, Chunk> incoming = new LinkedHashMap<>();
List<Chunk> unchanged = new ArrayList<>();
List<Chunk> contentChanged = new ArrayList<>();
List<Chunk> unknownIds = new ArrayList<>();
}
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
SyncState state = new SyncState();
for (Chunk c : latestChunks) {
state.incoming.put(c.pointId, c);
}
Map<String, String> stored = new HashMap<>();
var points = client.retrieveAsync(
COLLECTION,
state.incoming.keySet().stream()
.map(pid -> id(UUID.fromString(pid)))
.collect(Collectors.toList()),
WithPayloadSelectorFactory.include(List.of("content_hash")),
WithVectorsSelectorFactory.enable(false),
null).get();
for (var p : points) {
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
}
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
String pid = e.getKey();
Chunk c = e.getValue();
if (c.contentHash.equals(stored.get(pid))) {
state.unchanged.add(c);
} else if (stored.containsKey(pid)) {
state.contentChanged.add(c);
} else {
state.unknownIds.add(c);
}
}
return state;
}
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
if (contentChanged.isEmpty()) {
return;
}
List<PointStruct> points = new ArrayList<>();
for (Chunk c : contentChanged) {
points.add(
PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build());
}
client.upsertAsync(COLLECTION, points).get();
}
// Reuse an existing embedding when the same text is already stored; embed only what is new.
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
int reused = 0;
int added = 0;
for (Chunk c : unknownIds) {
Filter sameText = Filter.newBuilder()
.addMust(matchKeyword("content_hash", c.contentHash))
.build();
var hits = client.scrollAsync(
ScrollPoints.newBuilder()
.setCollectionName(COLLECTION)
.setFilter(sameText)
.setLimit(1)
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
.setWithVectors(WithVectorsSelectorFactory.enable(true))
.build()).get().getResultList();
PointStruct point;
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
point = PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(vectors(vector(
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
.getDataList())))
.putAllPayload(
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
.build();
reused++;
} else { // genuinely new content: embed and insert
point = PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build();
added++;
}
client.upsertAsync(COLLECTION, List.of(point)).get();
}
return new int[] {reused, added};
}
// Remove every point the current crawl no longer contains. Returns how many.
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
if (incomingIds.isEmpty()) {
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
}
Filter stale = Filter.newBuilder()
.addMustNot(hasId(
incomingIds.keySet().stream()
.map(pid -> id(UUID.fromString(pid)))
.collect(Collectors.toList())))
.build();
long toDelete = client.countAsync(COLLECTION, stale, true).get();
// potential check against a threshold to avoid accidental mass deletion could be added here
client.deleteAsync(COLLECTION, stale).get();
return toDelete;
}
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
checkGate(); // refuse to mix embedding models or pipeline versions
List<Chunk> chunks = prepareChunksForSync(latestChunks);
SyncState state = splitByState(chunks);
reEmbedChanged(state.contentChanged);
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
long deleted = deleteGone(state.incoming);
return Map.of(
"unchanged", (long) state.unchanged.size(),
"re-embedded", (long) state.contentChanged.size(),
"reused_embedding", (long) reusedAdded[0],
"added", (long) reusedAdded[1],
"deleted", deleted);
}
static void runSync() throws Exception {
Map<String, Long> run = sync(LATEST_CHUNKS);
System.out.println(run);
}
```
@@ -0,0 +1,4 @@
```csharp
foreach (var field in new[] { "content_hash", "url", "section_url" })
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
```
@@ -0,0 +1,9 @@
```go
for _, field := range []string{"content_hash", "url", "section_url"} {
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
CollectionName: COLLECTION,
FieldName: field,
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
})
}
```
@@ -0,0 +1,8 @@
```java
static void createPayloadIndexes() throws Exception {
for (String field : List.of("content_hash", "url", "section_url")) {
client.createPayloadIndexAsync(
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
}
}
```
@@ -0,0 +1,4 @@
```python
for field in ("content_hash", "url", "section_url"):
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
```
@@ -0,0 +1,11 @@
```rust
for field in ["content_hash", "url", "section_url"] {
client
.create_field_index(CreateFieldIndexCollectionBuilder::new(
COLLECTION,
field,
FieldType::Keyword,
))
.await?;
}
```
@@ -0,0 +1,8 @@
```typescript
for (const field of ["content_hash", "url", "section_url"]) {
await client.createPayloadIndex(COLLECTION, {
field_name: field,
field_schema: "keyword",
});
}
```
@@ -0,0 +1,12 @@
```csharp
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
{
["url"] = chunk.Url,
["anchor"] = chunk.Anchor,
["chunk_num"] = chunk.ChunkNum,
["section_url"] = chunk.SectionUrl,
["text"] = chunk.Text,
["content_hash"] = chunk.ContentHash,
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
};
```
@@ -0,0 +1,16 @@
```go
payload := func(c Chunk, lastUpdated string) map[string]any {
if lastUpdated == "" {
lastUpdated = time.Now().UTC().Format(time.RFC3339)
}
return map[string]any{
"url": c.URL,
"anchor": c.Anchor,
"chunk_num": c.ChunkNum,
"section_url": c.SectionURL,
"text": c.Text,
"content_hash": c.ContentHash,
"last_updated": lastUpdated,
}
}
```
@@ -0,0 +1,15 @@
```java
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
Map<String, Value> p = new HashMap<>();
p.put("url", value(chunk.url));
p.put("anchor", value(chunk.anchor));
p.put("chunk_num", value(chunk.chunkNum));
p.put("section_url", value(chunk.sectionUrl));
p.put("text", value(chunk.text));
p.put("content_hash", value(chunk.contentHash));
p.put("last_updated", value(lastUpdated != null
? lastUpdated
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
return p;
}
```
@@ -0,0 +1,12 @@
```python
def payload(chunk, last_updated=None):
return {
"url": chunk["url"],
"anchor": chunk["anchor"],
"chunk_num": chunk["chunk_num"],
"section_url": chunk["section_url"],
"text": chunk["text"],
"content_hash": chunk["content_hash"],
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
}
```
@@ -0,0 +1,16 @@
```rust
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
let last_updated = last_updated.unwrap_or_else(|| {
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
});
Ok(Payload::try_from(serde_json::json!({
"url": chunk.url,
"anchor": chunk.anchor,
"chunk_num": chunk.chunk_num,
"section_url": chunk.section_url,
"text": chunk.text,
"content_hash": chunk.content_hash,
"last_updated": last_updated,
}))?)
}
```
@@ -0,0 +1,13 @@
```typescript
function payload(chunk: SyncChunk, lastUpdated?: string) {
return {
url: chunk.url,
anchor: chunk.anchor,
chunk_num: chunk.chunk_num,
section_url: chunk.section_url,
text: chunk.text,
content_hash: chunk.content_hash,
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
};
}
```
@@ -0,0 +1,12 @@
```csharp
await client.UpsertAsync(
collectionName: COLLECTION,
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
}).ToList(),
wait: true
);
```
@@ -0,0 +1,16 @@
```go
var points []*qdrant.PointStruct
for _, c := range prepareChunksForSync(CHUNKS) {
points = append(points, &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
})
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: points,
Wait: qdrant.PtrOf(true),
})
```
@@ -0,0 +1,21 @@
```java
static void populate() throws Exception {
List<PointStruct> points = new ArrayList<>();
for (Chunk c : prepareChunksForSync(CHUNKS)) {
points.add(
PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build());
}
client.upsertAsync(COLLECTION, points).get();
}
```
@@ -0,0 +1,10 @@
```python
client.upsert(COLLECTION, points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
for c in prepare_chunks_for_sync(CHUNKS)
], wait=True)
```
@@ -0,0 +1,16 @@
```rust
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
.iter()
.map(|c| {
Ok(PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
))
})
.collect::<anyhow::Result<_>>()?;
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
.await?;
```
@@ -0,0 +1,10 @@
```typescript
await client.upsert(COLLECTION, {
points: prepareChunksForSync(CHUNKS).map((c) => ({
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
})),
wait: true,
});
```
@@ -0,0 +1,203 @@
```python
import os
from qdrant_client import QdrantClient, models
QDRANT_URL = os.getenv("QDRANT_URL")
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
client = QdrantClient(
url=QDRANT_URL,
api_key=QDRANT_API_KEY,
cloud_inference=True
)
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
PIPELINE = "docs-prep-pipeline-v1"
COLLECTION = "docs-sync-tutorial"
client.create_collection(
COLLECTION,
vectors_config=models.VectorParams(
size=384, # all-MiniLM-L6-v2 output dimension
distance=models.Distance.COSINE,
),
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
)
def check_gate():
# compare this pipeline's constants against what the collection records about itself
meta = client.get_collection(COLLECTION).config.metadata or {}
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
import hashlib
import uuid
from datetime import datetime, timezone
def content_hash(text):
return hashlib.sha256(text.encode()).hexdigest()
def point_id(url, anchor, num):
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
def prepare_chunks_for_sync(chunks):
"""Derive both values (and the section address) for every raw chunk."""
out = []
for c in chunks:
text = normalize(c["text"])
out.append({
**c,
"text": text,
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
"content_hash": content_hash(text),
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
})
return out
def payload(chunk, last_updated=None):
return {
"url": chunk["url"],
"anchor": chunk["anchor"],
"chunk_num": chunk["chunk_num"],
"section_url": chunk["section_url"],
"text": chunk["text"],
"content_hash": chunk["content_hash"],
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
}
for field in ("content_hash", "url", "section_url"):
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
client.upsert(COLLECTION, points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
for c in prepare_chunks_for_sync(CHUNKS)
], wait=True)
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
client.query_points(
COLLECTION,
query=models.Document(text=QUERY, model=MODEL),
limit=3,
with_payload=["section_url", "text"],
)
def split_by_state(latest_chunks):
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
incoming = {c["point_id"]: c for c in latest_chunks}
stored = {}
points = client.retrieve(
COLLECTION,
ids=list(incoming),
with_payload=["content_hash"],
with_vectors=False,
)
for p in points:
stored[str(p.id)] = p.payload["content_hash"]
unchanged, content_changed, unknown_ids = [], [], []
for pid, c in incoming.items():
if stored.get(pid) == c["content_hash"]:
unchanged.append(c)
elif pid in stored:
content_changed.append(c)
else:
unknown_ids.append(c)
return incoming, unchanged, content_changed, unknown_ids
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
def re_embed_changed(content_changed):
if not content_changed:
return
client.upsert(COLLECTION,
points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
for c in content_changed],
wait=True)
def reuse_or_add(unknown_ids):
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
reused, added = 0, 0
for c in unknown_ids:
same_text = models.Filter(must=[
models.FieldCondition(
key="content_hash",
match=models.MatchValue(value=c["content_hash"]),
)
])
hits, _ = client.scroll(
COLLECTION,
scroll_filter=same_text,
limit=1,
with_payload=["last_updated"],
with_vectors=True,
)
if hits: # same text, new address: copy the vector, keep its last_updated
point = models.PointStruct(
id=c["point_id"],
vector=hits[0].vector,
payload=payload(c, hits[0].payload["last_updated"]),
)
reused += 1
else: # genuinely new content: embed and insert
point = models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
added += 1
client.upsert(COLLECTION, points=[point], wait=True)
return reused, added
def delete_gone(incoming_ids):
"""Remove every point the current crawl no longer contains. Returns how many."""
if not incoming_ids:
raise ValueError("Refusing to delete from an empty source snapshot.")
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
to_delete = client.count(COLLECTION, count_filter=stale).count
# potential check against a threshold to avoid accidental mass deletion could be added here
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
return to_delete
def sync(latest_chunks):
check_gate() # refuse to mix embedding models or pipeline versions
chunks = prepare_chunks_for_sync(latest_chunks)
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
re_embed_changed(content_changed)
reused, added = reuse_or_add(unknown_ids)
deleted = delete_gone(incoming_ids)
return {
"unchanged": len(unchanged),
"re-embedded": len(content_changed),
"reused_embedding": reused,
"added": added,
"deleted": deleted,
}
run = sync(LATEST_CHUNKS)
print(run)
```
@@ -0,0 +1,17 @@
```csharp
async Task ReEmbedChanged(List<Chunk> contentChanged)
{
if (contentChanged.Count == 0)
return;
await client.UpsertAsync(
collectionName: COLLECTION,
points: contentChanged.Select(c => new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
}).ToList(),
wait: true
);
}
```
@@ -0,0 +1,20 @@
```go
reEmbedChanged := func(contentChanged []Chunk) {
if len(contentChanged) == 0 {
return
}
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
for _, c := range contentChanged {
points = append(points, &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
})
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: points,
Wait: qdrant.PtrOf(true),
})
}
```
@@ -0,0 +1,23 @@
```java
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
if (contentChanged.isEmpty()) {
return;
}
List<PointStruct> points = new ArrayList<>();
for (Chunk c : contentChanged) {
points.add(
PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build());
}
client.upsertAsync(COLLECTION, points).get();
}
```
@@ -0,0 +1,14 @@
```python
def re_embed_changed(content_changed):
if not content_changed:
return
client.upsert(COLLECTION,
points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
for c in content_changed],
wait=True)
```
@@ -0,0 +1,22 @@
```rust
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
if content_changed.is_empty() {
return Ok(());
}
let points: Vec<PointStruct> = content_changed
.iter()
.map(|c| {
Ok(PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
))
})
.collect::<anyhow::Result<_>>()?;
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
.await?;
Ok(())
}
```
@@ -0,0 +1,15 @@
```typescript
async function reEmbedChanged(contentChanged: SyncChunk[]) {
if (contentChanged.length === 0) {
return;
}
await client.upsert(COLLECTION, {
points: contentChanged.map((c) => ({
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
})),
wait: true,
});
}
```
@@ -0,0 +1,48 @@
```csharp
// Reuse an existing embedding when the same text is already stored; embed only what is new.
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
{
int reused = 0, added = 0;
foreach (var c in unknownIds)
{
var sameText = new Filter
{
Must = { MatchKeyword("content_hash", c.ContentHash) }
};
var hits = (await client.ScrollAsync(
COLLECTION,
filter: sameText,
limit: 1,
payloadSelector: new[] { "last_updated" },
vectorsSelector: true
)).Result;
PointStruct point;
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
{
point = new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
};
reused++;
}
else // genuinely new content: embed and insert
{
point = new PointStruct
{
Id = new PointId { Uuid = c.PointId },
Vectors = new Document { Text = c.Text, Model = MODEL },
Payload = { Payload(c) },
};
added++;
}
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
}
return (reused, added);
}
```
@@ -0,0 +1,46 @@
```go
// reuse an existing embedding when the same text is already stored; embed only what is new
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
reused, added := 0, 0
for _, c := range unknownIDs {
sameText := &qdrant.Filter{
Must: []*qdrant.Condition{
qdrant.NewMatch("content_hash", c.ContentHash),
},
}
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
CollectionName: COLLECTION,
Filter: sameText,
Limit: qdrant.PtrOf(uint32(1)),
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
WithVectors: qdrant.NewWithVectors(true),
})
var point *qdrant.PointStruct
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
point = &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
}
reused++
} else { // genuinely new content: embed and insert
point = &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
}
added++
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: []*qdrant.PointStruct{point},
Wait: qdrant.PtrOf(true),
})
}
return reused, added
}
```
@@ -0,0 +1,52 @@
```java
// Reuse an existing embedding when the same text is already stored; embed only what is new.
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
int reused = 0;
int added = 0;
for (Chunk c : unknownIds) {
Filter sameText = Filter.newBuilder()
.addMust(matchKeyword("content_hash", c.contentHash))
.build();
var hits = client.scrollAsync(
ScrollPoints.newBuilder()
.setCollectionName(COLLECTION)
.setFilter(sameText)
.setLimit(1)
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
.setWithVectors(WithVectorsSelectorFactory.enable(true))
.build()).get().getResultList();
PointStruct point;
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
point = PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(vectors(vector(
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
.getDataList())))
.putAllPayload(
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
.build();
reused++;
} else { // genuinely new content: embed and insert
point = PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build();
added++;
}
client.upsertAsync(COLLECTION, List.of(point)).get();
}
return new int[] {reused, added};
}
```
@@ -0,0 +1,39 @@
```python
def reuse_or_add(unknown_ids):
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
reused, added = 0, 0
for c in unknown_ids:
same_text = models.Filter(must=[
models.FieldCondition(
key="content_hash",
match=models.MatchValue(value=c["content_hash"]),
)
])
hits, _ = client.scroll(
COLLECTION,
scroll_filter=same_text,
limit=1,
with_payload=["last_updated"],
with_vectors=True,
)
if hits: # same text, new address: copy the vector, keep its last_updated
point = models.PointStruct(
id=c["point_id"],
vector=hits[0].vector,
payload=payload(c, hits[0].payload["last_updated"]),
)
reused += 1
else: # genuinely new content: embed and insert
point = models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
added += 1
client.upsert(COLLECTION, points=[point], wait=True)
return reused, added
```
@@ -0,0 +1,51 @@
```rust
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
let (mut reused, mut added) = (0, 0);
for c in unknown_ids {
let same_text =
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
let hits = client
.scroll(
ScrollPointsBuilder::new(COLLECTION)
.filter(same_text)
.limit(1)
.with_payload(PayloadIncludeSelector::new(vec![
"last_updated".to_string()
]))
.with_vectors(true),
)
.await?
.result;
let point = if let Some(hit) = hits.into_iter().next() {
// same text, new address: copy the vector, keep its last_updated
let last_updated = hit.get("last_updated").as_str().cloned();
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
Some(vector_output::Vector::Dense(dense)) => dense.data,
_ => anyhow::bail!("expected a dense vector on the stored point"),
},
_ => anyhow::bail!("expected a dense vector on the stored point"),
};
reused += 1;
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
} else {
// genuinely new content: embed and insert
added += 1;
PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
)
};
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
.await?;
}
Ok((reused, added))
}
```
@@ -0,0 +1,45 @@
```typescript
// Reuse an existing embedding when the same text is already stored; embed only what is new.
async function reuseOrAdd(unknownIds: SyncChunk[]) {
let reused = 0;
let added = 0;
for (const c of unknownIds) {
const sameText = {
must: [
{
key: "content_hash",
match: { value: c.content_hash },
},
],
};
const hits = (await client.scroll(COLLECTION, {
filter: sameText,
limit: 1,
with_payload: ["last_updated"],
with_vector: true,
})).points;
let point: Schemas["PointStruct"];
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
point = {
id: c.point_id,
vector: hits[0].vector as number[],
payload: payload(c, hits[0].payload?.last_updated as string),
};
reused += 1;
} else { // genuinely new content: embed and insert
point = {
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
};
added += 1;
}
await client.upsert(COLLECTION, { points: [point], wait: true });
}
return { reused, added };
}
```
@@ -0,0 +1,5 @@
```csharp
var run = await Sync(LATEST_CHUNKS);
foreach (var (op, count) in run)
Console.WriteLine($"{op}: {count}");
```
@@ -0,0 +1,4 @@
```go
run := sync(LATEST_CHUNKS)
fmt.Println(run)
```
@@ -0,0 +1,6 @@
```java
static void runSync() throws Exception {
Map<String, Long> run = sync(LATEST_CHUNKS);
System.out.println(run);
}
```
@@ -0,0 +1,4 @@
```python
run = sync(LATEST_CHUNKS)
print(run)
```
@@ -0,0 +1,4 @@
```rust
let run = sync(&client, &latest_chunks).await?;
println!("{run:?}");
```
@@ -0,0 +1,4 @@
```typescript
const run = await sync(LATEST_CHUNKS);
console.log(run);
```
@@ -0,0 +1,322 @@
```rust
use serde_json::{json, Value};
use std::collections::HashMap;
use qdrant_client::qdrant::{
point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
};
use qdrant_client::{Payload, Qdrant};
use sha2::{Digest, Sha256};
let qdrant_url = std::env::var("QDRANT_URL")?;
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
let client = Qdrant::from_url(&qdrant_url)
.api_key(qdrant_api_key)
.build()?;
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
const PIPELINE: &str = "docs-prep-pipeline-v1";
const COLLECTION: &str = "docs-sync-tutorial";
let mut metadata: HashMap<String, Value> = HashMap::new();
metadata.insert("embedding_model".to_string(), json!(MODEL));
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
client
.create_collection(
CreateCollectionBuilder::new(COLLECTION)
.vectors_config(VectorParamsBuilder::new(
384, // all-MiniLM-L6-v2 output dimension
Distance::Cosine,
))
.metadata(metadata),
)
.await?;
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
// compare this pipeline's constants against what the collection records about itself
let meta = client
.collection_info(COLLECTION)
.await?
.result
.and_then(|info| info.config)
.map(|config| config.metadata)
.unwrap_or_default();
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
!= Some(PIPELINE)
{
anyhow::bail!(
"collection was built by {meta:?}: full re-embed into a fresh collection required"
);
}
Ok(())
}
fn content_hash(text: &str) -> String {
Sha256::digest(text.as_bytes())
.iter()
.map(|byte| format!("{byte:02x}"))
.collect()
}
fn point_id(url: &str, anchor: &str, num: u32) -> String {
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
uuid::Uuid::new_v5(
&uuid::Uuid::NAMESPACE_URL,
format!("{url}#{anchor}::{num}").as_bytes(),
)
.to_string()
}
/// Derive both values (and the section address) for every raw chunk.
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
chunks
.iter()
.map(|c| {
let text = normalize(&c.text);
Chunk {
text: text.clone(),
section_url: if c.anchor.is_empty() {
c.url.clone()
} else {
format!("{}#{}", c.url, c.anchor)
},
content_hash: content_hash(&text),
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
..c.clone()
}
})
.collect()
}
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
let last_updated = last_updated.unwrap_or_else(|| {
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
});
Ok(Payload::try_from(serde_json::json!({
"url": chunk.url,
"anchor": chunk.anchor,
"chunk_num": chunk.chunk_num,
"section_url": chunk.section_url,
"text": chunk.text,
"content_hash": chunk.content_hash,
"last_updated": last_updated,
}))?)
}
for field in ["content_hash", "url", "section_url"] {
client
.create_field_index(CreateFieldIndexCollectionBuilder::new(
COLLECTION,
field,
FieldType::Keyword,
))
.await?;
}
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
.iter()
.map(|c| {
Ok(PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
))
})
.collect::<anyhow::Result<_>>()?;
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
.await?;
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
client
.query(
QueryPointsBuilder::new(COLLECTION)
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
.limit(3)
.with_payload(PayloadIncludeSelector::new(vec![
"section_url".to_string(),
"text".to_string(),
])),
)
.await?;
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async fn split_by_state(
client: &Qdrant,
latest_chunks: &[Chunk],
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
let incoming: HashMap<String, Chunk> = latest_chunks
.iter()
.map(|c| (c.point_id.clone(), c.clone()))
.collect();
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
let points = client
.get_points(
GetPointsBuilder::new(COLLECTION, ids)
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
.with_vectors(false),
)
.await?;
let mut stored: HashMap<String, String> = HashMap::new();
for p in points.result {
let hash = p.get("content_hash").as_str().cloned();
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
(p.id.and_then(|i| i.point_id_options), hash)
{
stored.insert(id, hash);
}
}
let (mut unchanged, mut content_changed, mut unknown_ids) =
(Vec::new(), Vec::new(), Vec::new());
for (pid, c) in &incoming {
if stored.get(pid) == Some(&c.content_hash) {
unchanged.push(c.clone());
} else if stored.contains_key(pid) {
content_changed.push(c.clone());
} else {
unknown_ids.push(c.clone());
}
}
Ok((incoming, unchanged, content_changed, unknown_ids))
}
let (incoming_ids, unchanged, content_changed, unknown_ids) =
split_by_state(&client, &latest_chunks).await?;
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
if content_changed.is_empty() {
return Ok(());
}
let points: Vec<PointStruct> = content_changed
.iter()
.map(|c| {
Ok(PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
))
})
.collect::<anyhow::Result<_>>()?;
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
.await?;
Ok(())
}
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
let (mut reused, mut added) = (0, 0);
for c in unknown_ids {
let same_text =
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
let hits = client
.scroll(
ScrollPointsBuilder::new(COLLECTION)
.filter(same_text)
.limit(1)
.with_payload(PayloadIncludeSelector::new(vec![
"last_updated".to_string()
]))
.with_vectors(true),
)
.await?
.result;
let point = if let Some(hit) = hits.into_iter().next() {
// same text, new address: copy the vector, keep its last_updated
let last_updated = hit.get("last_updated").as_str().cloned();
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
Some(vector_output::Vector::Dense(dense)) => dense.data,
_ => anyhow::bail!("expected a dense vector on the stored point"),
},
_ => anyhow::bail!("expected a dense vector on the stored point"),
};
reused += 1;
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
} else {
// genuinely new content: embed and insert
added += 1;
PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
)
};
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
.await?;
}
Ok((reused, added))
}
/// Remove every point the current crawl no longer contains. Returns how many.
async fn delete_gone(
client: &Qdrant,
incoming_ids: &HashMap<String, Chunk>,
) -> anyhow::Result<u64> {
if incoming_ids.is_empty() {
anyhow::bail!("Refusing to delete from an empty source snapshot.");
}
let stale = Filter::must_not([Condition::has_id(
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
)]);
let to_delete = client
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
.await?
.result
.map(|r| r.count)
.unwrap_or(0);
// potential check against a threshold to avoid accidental mass deletion could be added here
client
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
.await?;
Ok(to_delete)
}
async fn sync(
client: &Qdrant,
latest_chunks: &[Chunk],
) -> anyhow::Result<HashMap<&'static str, usize>> {
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
let chunks = prepare_chunks_for_sync(latest_chunks);
let (incoming_ids, unchanged, content_changed, unknown_ids) =
split_by_state(client, &chunks).await?;
re_embed_changed(client, &content_changed).await?;
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
let deleted = delete_gone(client, &incoming_ids).await?;
Ok(HashMap::from([
("unchanged", unchanged.len()),
("re-embedded", content_changed.len()),
("reused_embedding", reused),
("added", added),
("deleted", deleted as usize),
]))
}
let run = sync(&client, &latest_chunks).await?;
println!("{run:?}");
```
@@ -0,0 +1,10 @@
```csharp
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
await client.QueryAsync(
collectionName: COLLECTION,
query: new Document { Text = QUERY, Model = MODEL },
limit: 3,
payloadSelector: new[] { "section_url", "text" }
);
```
@@ -0,0 +1,10 @@
```go
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: COLLECTION,
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
Limit: qdrant.PtrOf(uint64(3)),
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
})
```
@@ -0,0 +1,19 @@
```java
static final String QUERY =
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
static void search() throws Exception {
client.queryAsync(
QueryPoints.newBuilder()
.setCollectionName(COLLECTION)
.setQuery(
nearest(
Document.newBuilder()
.setText(QUERY)
.setModel(MODEL)
.build()))
.setLimit(3)
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
.build()).get();
}
```
@@ -0,0 +1,10 @@
```python
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
client.query_points(
COLLECTION,
query=models.Document(text=QUERY, model=MODEL),
limit=3,
with_payload=["section_url", "text"],
)
```
@@ -0,0 +1,15 @@
```rust
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
client
.query(
QueryPointsBuilder::new(COLLECTION)
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
.limit(3)
.with_payload(PayloadIncludeSelector::new(vec![
"section_url".to_string(),
"text".to_string(),
])),
)
.await?;
```
@@ -0,0 +1,9 @@
```typescript
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
await client.query(COLLECTION, {
query: { text: QUERY, model: MODEL },
limit: 3,
with_payload: ["section_url", "text"],
});
```
@@ -0,0 +1,35 @@
```csharp
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
SplitByState(List<Chunk> latestChunks)
{
var incoming = latestChunks.ToDictionary(c => c.PointId);
var stored = new Dictionary<string, string>();
var points = await client.RetrieveAsync(
COLLECTION,
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
payloadSelector: new[] { "content_hash" },
vectorSelector: false
);
foreach (var p in points)
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
var unchanged = new List<Chunk>();
var contentChanged = new List<Chunk>();
var unknownIds = new List<Chunk>();
foreach (var (pid, c) in incoming)
{
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
unchanged.Add(c);
else if (stored.ContainsKey(pid))
contentChanged.Add(c);
else
unknownIds.Add(c);
}
return (incoming, unchanged, contentChanged, unknownIds);
}
var splitState = await SplitByState(LATEST_CHUNKS);
```
@@ -0,0 +1,39 @@
```go
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
incoming := make(map[string]Chunk, len(latestChunks))
ids := make([]*qdrant.PointId, 0, len(latestChunks))
for _, c := range latestChunks {
incoming[c.PointID] = c
ids = append(ids, qdrant.NewID(c.PointID))
}
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
CollectionName: COLLECTION,
Ids: ids,
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
WithVectors: qdrant.NewWithVectors(false),
})
stored := make(map[string]string, len(retrieved))
for _, p := range retrieved {
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
}
var unchanged, contentChanged, unknownIDs []Chunk
for pid, c := range incoming {
storedHash, found := stored[pid]
switch {
case found && storedHash == c.ContentHash:
unchanged = append(unchanged, c)
case found:
contentChanged = append(contentChanged, c)
default:
unknownIDs = append(unknownIDs, c)
}
}
return incoming, unchanged, contentChanged, unknownIDs
}
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
```
@@ -0,0 +1,43 @@
```java
static class SyncState {
Map<String, Chunk> incoming = new LinkedHashMap<>();
List<Chunk> unchanged = new ArrayList<>();
List<Chunk> contentChanged = new ArrayList<>();
List<Chunk> unknownIds = new ArrayList<>();
}
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
SyncState state = new SyncState();
for (Chunk c : latestChunks) {
state.incoming.put(c.pointId, c);
}
Map<String, String> stored = new HashMap<>();
var points = client.retrieveAsync(
COLLECTION,
state.incoming.keySet().stream()
.map(pid -> id(UUID.fromString(pid)))
.collect(Collectors.toList()),
WithPayloadSelectorFactory.include(List.of("content_hash")),
WithVectorsSelectorFactory.enable(false),
null).get();
for (var p : points) {
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
}
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
String pid = e.getKey();
Chunk c = e.getValue();
if (c.contentHash.equals(stored.get(pid))) {
state.unchanged.add(c);
} else if (stored.containsKey(pid)) {
state.contentChanged.add(c);
} else {
state.unknownIds.add(c);
}
}
return state;
}
```
@@ -0,0 +1,28 @@
```python
def split_by_state(latest_chunks):
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
incoming = {c["point_id"]: c for c in latest_chunks}
stored = {}
points = client.retrieve(
COLLECTION,
ids=list(incoming),
with_payload=["content_hash"],
with_vectors=False,
)
for p in points:
stored[str(p.id)] = p.payload["content_hash"]
unchanged, content_changed, unknown_ids = [], [], []
for pid, c in incoming.items():
if stored.get(pid) == c["content_hash"]:
unchanged.append(c)
elif pid in stored:
content_changed.append(c)
else:
unknown_ids.append(c)
return incoming, unchanged, content_changed, unknown_ids
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
```
@@ -0,0 +1,48 @@
```rust
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async fn split_by_state(
client: &Qdrant,
latest_chunks: &[Chunk],
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
let incoming: HashMap<String, Chunk> = latest_chunks
.iter()
.map(|c| (c.point_id.clone(), c.clone()))
.collect();
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
let points = client
.get_points(
GetPointsBuilder::new(COLLECTION, ids)
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
.with_vectors(false),
)
.await?;
let mut stored: HashMap<String, String> = HashMap::new();
for p in points.result {
let hash = p.get("content_hash").as_str().cloned();
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
(p.id.and_then(|i| i.point_id_options), hash)
{
stored.insert(id, hash);
}
}
let (mut unchanged, mut content_changed, mut unknown_ids) =
(Vec::new(), Vec::new(), Vec::new());
for (pid, c) in &incoming {
if stored.get(pid) == Some(&c.content_hash) {
unchanged.push(c.clone());
} else if stored.contains_key(pid) {
content_changed.push(c.clone());
} else {
unknown_ids.push(c.clone());
}
}
Ok((incoming, unchanged, content_changed, unknown_ids))
}
let (incoming_ids, unchanged, content_changed, unknown_ids) =
split_by_state(&client, &latest_chunks).await?;
```
@@ -0,0 +1,33 @@
```typescript
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async function splitByState(latestChunks: SyncChunk[]) {
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
const stored = new Map<string, string>();
const points = await client.retrieve(COLLECTION, {
ids: [...incoming.keys()],
with_payload: ["content_hash"],
with_vector: false,
});
for (const p of points) {
stored.set(String(p.id), p.payload?.content_hash as string);
}
const unchanged: SyncChunk[] = [];
const contentChanged: SyncChunk[] = [];
const unknownIds: SyncChunk[] = [];
for (const [pid, c] of incoming) {
if (stored.get(pid) === c.content_hash) {
unchanged.push(c);
} else if (stored.has(pid)) {
contentChanged.push(c);
} else {
unknownIds.push(c);
}
}
return { incoming, unchanged, contentChanged, unknownIds };
}
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
```
@@ -0,0 +1,22 @@
```csharp
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
{
await CheckGate(); // refuse to mix embedding models or pipeline versions
var chunks = PrepareChunksForSync(latestChunks);
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
await ReEmbedChanged(contentChanged);
var (reused, added) = await ReuseOrAdd(unknownIds);
var deleted = await DeleteGone(incomingIds);
return new Dictionary<string, long>
{
["unchanged"] = unchanged.Count,
["re-embedded"] = contentChanged.Count,
["reused_embedding"] = reused,
["added"] = added,
["deleted"] = (long)deleted,
};
}
```
@@ -0,0 +1,20 @@
```go
sync := func(latestChunks []Chunk) map[string]int {
checkGate() // refuse to mix embedding models or pipeline versions
chunks := prepareChunksForSync(latestChunks)
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
reEmbedChanged(contentChanged)
reused, added := reuseOrAdd(unknownIDs)
deleted := deleteGone(incomingIDs)
return map[string]int{
"unchanged": len(unchanged),
"re-embedded": len(contentChanged),
"reused_embedding": reused,
"added": added,
"deleted": deleted,
}
}
```
@@ -0,0 +1,19 @@
```java
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
checkGate(); // refuse to mix embedding models or pipeline versions
List<Chunk> chunks = prepareChunksForSync(latestChunks);
SyncState state = splitByState(chunks);
reEmbedChanged(state.contentChanged);
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
long deleted = deleteGone(state.incoming);
return Map.of(
"unchanged", (long) state.unchanged.size(),
"re-embedded", (long) state.contentChanged.size(),
"reused_embedding", (long) reusedAdded[0],
"added", (long) reusedAdded[1],
"deleted", deleted);
}
```
@@ -0,0 +1,19 @@
```python
def sync(latest_chunks):
check_gate() # refuse to mix embedding models or pipeline versions
chunks = prepare_chunks_for_sync(latest_chunks)
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
re_embed_changed(content_changed)
reused, added = reuse_or_add(unknown_ids)
deleted = delete_gone(incoming_ids)
return {
"unchanged": len(unchanged),
"re-embedded": len(content_changed),
"reused_embedding": reused,
"added": added,
"deleted": deleted,
}
```
@@ -0,0 +1,24 @@
```rust
async fn sync(
client: &Qdrant,
latest_chunks: &[Chunk],
) -> anyhow::Result<HashMap<&'static str, usize>> {
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
let chunks = prepare_chunks_for_sync(latest_chunks);
let (incoming_ids, unchanged, content_changed, unknown_ids) =
split_by_state(client, &chunks).await?;
re_embed_changed(client, &content_changed).await?;
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
let deleted = delete_gone(client, &incoming_ids).await?;
Ok(HashMap::from([
("unchanged", unchanged.len()),
("re-embedded", content_changed.len()),
("reused_embedding", reused),
("added", added),
("deleted", deleted as usize),
]))
}
```
@@ -0,0 +1,20 @@
```typescript
async function sync(latestChunks: RawChunk[]) {
await checkGate(); // refuse to mix embedding models or pipeline versions
const chunks = prepareChunksForSync(latestChunks);
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
await reEmbedChanged(contentChanged);
const { reused, added } = await reuseOrAdd(unknownIds);
const deleted = await deleteGone(incoming);
return {
"unchanged": unchanged.length,
"re-embedded": contentChanged.length,
"reused_embedding": reused,
"added": added,
"deleted": deleted,
};
}
```
@@ -0,0 +1,230 @@
```typescript
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
const QDRANT_URL = process.env.QDRANT_URL;
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
const client = new QdrantClient({
url: QDRANT_URL,
apiKey: QDRANT_API_KEY,
});
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
const PIPELINE = "docs-prep-pipeline-v1";
const COLLECTION = "docs-sync-tutorial";
await client.createCollection(COLLECTION, {
vectors: {
size: 384, // all-MiniLM-L6-v2 output dimension
distance: "Cosine",
},
});
await client.updateCollection(COLLECTION, {
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
});
async function checkGate() {
// compare this pipeline's constants against what the collection records about itself
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
{}) as Record<string, unknown>;
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
}
}
import { createHash } from "node:crypto";
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
function contentHash(text: string): string {
return createHash("sha256").update(text).digest("hex");
}
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
function pointId(url: string, anchor: string, num: number): string {
// Qdrant accepts any well-formed UUID as a point ID:
// hash the address, format the digest as a UUID, and the same address always yields the same ID
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
}
// Derive both values (and the section address) for every raw chunk.
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
return chunks.map((c) => {
const text = normalize(c.text);
return {
...c,
text,
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
content_hash: contentHash(text),
point_id: pointId(c.url, c.anchor, c.chunk_num),
};
});
}
function payload(chunk: SyncChunk, lastUpdated?: string) {
return {
url: chunk.url,
anchor: chunk.anchor,
chunk_num: chunk.chunk_num,
section_url: chunk.section_url,
text: chunk.text,
content_hash: chunk.content_hash,
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
};
}
for (const field of ["content_hash", "url", "section_url"]) {
await client.createPayloadIndex(COLLECTION, {
field_name: field,
field_schema: "keyword",
});
}
await client.upsert(COLLECTION, {
points: prepareChunksForSync(CHUNKS).map((c) => ({
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
})),
wait: true,
});
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
await client.query(COLLECTION, {
query: { text: QUERY, model: MODEL },
limit: 3,
with_payload: ["section_url", "text"],
});
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async function splitByState(latestChunks: SyncChunk[]) {
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
const stored = new Map<string, string>();
const points = await client.retrieve(COLLECTION, {
ids: [...incoming.keys()],
with_payload: ["content_hash"],
with_vector: false,
});
for (const p of points) {
stored.set(String(p.id), p.payload?.content_hash as string);
}
const unchanged: SyncChunk[] = [];
const contentChanged: SyncChunk[] = [];
const unknownIds: SyncChunk[] = [];
for (const [pid, c] of incoming) {
if (stored.get(pid) === c.content_hash) {
unchanged.push(c);
} else if (stored.has(pid)) {
contentChanged.push(c);
} else {
unknownIds.push(c);
}
}
return { incoming, unchanged, contentChanged, unknownIds };
}
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
async function reEmbedChanged(contentChanged: SyncChunk[]) {
if (contentChanged.length === 0) {
return;
}
await client.upsert(COLLECTION, {
points: contentChanged.map((c) => ({
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
})),
wait: true,
});
}
// Reuse an existing embedding when the same text is already stored; embed only what is new.
async function reuseOrAdd(unknownIds: SyncChunk[]) {
let reused = 0;
let added = 0;
for (const c of unknownIds) {
const sameText = {
must: [
{
key: "content_hash",
match: { value: c.content_hash },
},
],
};
const hits = (await client.scroll(COLLECTION, {
filter: sameText,
limit: 1,
with_payload: ["last_updated"],
with_vector: true,
})).points;
let point: Schemas["PointStruct"];
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
point = {
id: c.point_id,
vector: hits[0].vector as number[],
payload: payload(c, hits[0].payload?.last_updated as string),
};
reused += 1;
} else { // genuinely new content: embed and insert
point = {
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
};
added += 1;
}
await client.upsert(COLLECTION, { points: [point], wait: true });
}
return { reused, added };
}
// Remove every point the current crawl no longer contains. Returns how many.
async function deleteGone(incoming: Map<string, SyncChunk>) {
if (incoming.size === 0) {
throw new Error("Refusing to delete from an empty source snapshot.");
}
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
// potential check against a threshold to avoid accidental mass deletion could be added here
await client.delete(COLLECTION, { filter: stale, wait: true });
return toDelete;
}
async function sync(latestChunks: RawChunk[]) {
await checkGate(); // refuse to mix embedding models or pipeline versions
const chunks = prepareChunksForSync(latestChunks);
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
await reEmbedChanged(contentChanged);
const { reused, added } = await reuseOrAdd(unknownIds);
const deleted = await deleteGone(incoming);
return {
"unchanged": unchanged.length,
"re-embedded": contentChanged.length,
"reused_embedding": reused,
"added": added,
"deleted": deleted,
};
}
const run = await sync(LATEST_CHUNKS);
console.log(run);
```
@@ -0,0 +1,359 @@
package snippet
import (
"context"
"crypto/sha256"
"encoding/hex"
"fmt"
"os"
"regexp"
"strings"
"time"
"github.com/google/uuid"
"github.com/qdrant/go-client/qdrant"
)
func Main() {
// @block-start client-connection
QDRANT_URL := os.Getenv("QDRANT_URL")
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
client, err := qdrant.NewClient(&qdrant.Config{
Host: QDRANT_URL,
APIKey: QDRANT_API_KEY,
UseTLS: true,
})
// @block-end client-connection
// @hide-start
if err != nil {
panic(err)
}
// data and text normalization are not the lesson of this tutorial:
// the full CHUNKS list and normalize() live in the tutorial notebook
type Chunk struct {
URL string
Anchor string
ChunkNum int
Text string
SectionURL string
ContentHash string
PointID string
}
CHUNKS := []Chunk{
{
URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
Anchor: "prerequisites",
ChunkNum: 0,
Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
},
{
URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
Anchor: "step-3-enable-an-admin-api-key",
ChunkNum: 0,
Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
},
}
invisibleChars := regexp.MustCompile("[\u200B\u200C\u200D\uFEFF\u00AD]") // zero-width chars and soft hyphen
whitespace := regexp.MustCompile(`\s+`)
normalize := func(text string) string {
text = invisibleChars.ReplaceAllString(text, "")
return strings.TrimSpace(whitespace.ReplaceAllString(text, " "))
}
// @hide-end
// @block-start create-collection
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
PIPELINE := "docs-prep-pipeline-v1"
COLLECTION := "docs-sync-tutorial"
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
CollectionName: COLLECTION,
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
Size: 384, // all-MiniLM-L6-v2 output dimension
Distance: qdrant.Distance_Cosine,
}),
Metadata: qdrant.NewValueMap(map[string]any{
"embedding_model": MODEL,
"pipeline_version": PIPELINE,
}),
})
// @block-end create-collection
// @block-start check-gate
checkGate := func() {
// compare this pipeline's constants against what the collection records about itself
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
if err != nil { panic(err) } // @hide
meta := info.GetConfig().GetMetadata()
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
}
}
// @block-end check-gate
// @block-start identity-and-fingerprint
contentHash := func(text string) string {
sum := sha256.Sum256([]byte(text))
return hex.EncodeToString(sum[:])
}
pointID := func(url, anchor string, num int) string {
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
// marking the input as a URL-like name
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
}
// derive both values (and the section address) for every raw chunk
prepareChunksForSync := func(chunks []Chunk) []Chunk {
out := make([]Chunk, 0, len(chunks))
for _, c := range chunks {
c.Text = normalize(c.Text)
c.SectionURL = c.URL
if c.Anchor != "" {
c.SectionURL = c.URL + "#" + c.Anchor
}
c.ContentHash = contentHash(c.Text)
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
out = append(out, c)
}
return out
}
// @block-end identity-and-fingerprint
// @block-start payload
payload := func(c Chunk, lastUpdated string) map[string]any {
if lastUpdated == "" {
lastUpdated = time.Now().UTC().Format(time.RFC3339)
}
return map[string]any{
"url": c.URL,
"anchor": c.Anchor,
"chunk_num": c.ChunkNum,
"section_url": c.SectionURL,
"text": c.Text,
"content_hash": c.ContentHash,
"last_updated": lastUpdated,
}
}
// @block-end payload
// @block-start payload-indexes
for _, field := range []string{"content_hash", "url", "section_url"} {
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
CollectionName: COLLECTION,
FieldName: field,
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
})
}
// @block-end payload-indexes
// @block-start populate
var points []*qdrant.PointStruct
for _, c := range prepareChunksForSync(CHUNKS) {
points = append(points, &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
})
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: points,
Wait: qdrant.PtrOf(true),
})
// @block-end populate
// @block-start search
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: COLLECTION,
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
Limit: qdrant.PtrOf(uint64(3)),
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
})
// @block-end search
// @hide-start
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
LATEST_CHUNKS := prepareChunksForSync(CHUNKS)
// @hide-end
// @block-start split-by-state
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
incoming := make(map[string]Chunk, len(latestChunks))
ids := make([]*qdrant.PointId, 0, len(latestChunks))
for _, c := range latestChunks {
incoming[c.PointID] = c
ids = append(ids, qdrant.NewID(c.PointID))
}
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
CollectionName: COLLECTION,
Ids: ids,
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
WithVectors: qdrant.NewWithVectors(false),
})
if err != nil { panic(err) } // @hide
stored := make(map[string]string, len(retrieved))
for _, p := range retrieved {
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
}
var unchanged, contentChanged, unknownIDs []Chunk
for pid, c := range incoming {
storedHash, found := stored[pid]
switch {
case found && storedHash == c.ContentHash:
unchanged = append(unchanged, c)
case found:
contentChanged = append(contentChanged, c)
default:
unknownIDs = append(unknownIDs, c)
}
}
return incoming, unchanged, contentChanged, unknownIDs
}
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
// @block-end split-by-state
// @hide-start
_, _, _, _ = incomingIDs, unchanged, contentChanged, unknownIDs
// @hide-end
// @block-start re-embed-changed
reEmbedChanged := func(contentChanged []Chunk) {
if len(contentChanged) == 0 {
return
}
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
for _, c := range contentChanged {
points = append(points, &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
})
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: points,
Wait: qdrant.PtrOf(true),
})
}
// @block-end re-embed-changed
// @block-start reuse-or-add
// reuse an existing embedding when the same text is already stored; embed only what is new
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
reused, added := 0, 0
for _, c := range unknownIDs {
sameText := &qdrant.Filter{
Must: []*qdrant.Condition{
qdrant.NewMatch("content_hash", c.ContentHash),
},
}
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
CollectionName: COLLECTION,
Filter: sameText,
Limit: qdrant.PtrOf(uint32(1)),
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
WithVectors: qdrant.NewWithVectors(true),
})
if err != nil { panic(err) } // @hide
var point *qdrant.PointStruct
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
point = &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
}
reused++
} else { // genuinely new content: embed and insert
point = &qdrant.PointStruct{
Id: qdrant.NewID(c.PointID),
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
Payload: qdrant.NewValueMap(payload(c, "")),
}
added++
}
client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: COLLECTION,
Points: []*qdrant.PointStruct{point},
Wait: qdrant.PtrOf(true),
})
}
return reused, added
}
// @block-end reuse-or-add
// @block-start delete-gone
// remove every point the current crawl no longer contains, return how many
deleteGone := func(incomingIDs map[string]Chunk) int {
if len(incomingIDs) == 0 {
panic("Refusing to delete from an empty source snapshot.")
}
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
for pid := range incomingIDs {
ids = append(ids, qdrant.NewID(pid))
}
stale := &qdrant.Filter{
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
}
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
CollectionName: COLLECTION,
Filter: stale,
})
if err != nil { panic(err) } // @hide
// potential check against a threshold to avoid accidental mass deletion could be added here
client.Delete(context.Background(), &qdrant.DeletePoints{
CollectionName: COLLECTION,
Points: qdrant.NewPointsSelectorFilter(stale),
Wait: qdrant.PtrOf(true),
})
return int(toDelete)
}
// @block-end delete-gone
// @block-start sync
sync := func(latestChunks []Chunk) map[string]int {
checkGate() // refuse to mix embedding models or pipeline versions
chunks := prepareChunksForSync(latestChunks)
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
reEmbedChanged(contentChanged)
reused, added := reuseOrAdd(unknownIDs)
deleted := deleteGone(incomingIDs)
return map[string]int{
"unchanged": len(unchanged),
"re-embedded": len(contentChanged),
"reused_embedding": reused,
"added": added,
"deleted": deleted,
}
}
// @block-end sync
// @block-start run-sync
run := sync(LATEST_CHUNKS)
fmt.Println(run)
// @block-end run-sync
}
@@ -0,0 +1,415 @@
package com.example.snippets_amalgamation;
import static io.qdrant.client.ConditionFactory.hasId;
import static io.qdrant.client.ConditionFactory.matchKeyword;
import static io.qdrant.client.PointIdFactory.id;
import static io.qdrant.client.QueryFactory.nearest;
import static io.qdrant.client.ValueFactory.value;
import static io.qdrant.client.VectorFactory.vector;
import static io.qdrant.client.VectorsFactory.vectors;
import io.qdrant.client.QdrantClient;
import io.qdrant.client.QdrantGrpcClient;
import io.qdrant.client.VectorOutputHelper;
import io.qdrant.client.WithPayloadSelectorFactory;
import io.qdrant.client.WithVectorsSelectorFactory;
import io.qdrant.client.grpc.Collections.CreateCollection;
import io.qdrant.client.grpc.Collections.Distance;
import io.qdrant.client.grpc.Collections.PayloadSchemaType;
import io.qdrant.client.grpc.Collections.VectorParams;
import io.qdrant.client.grpc.Collections.VectorsConfig;
import io.qdrant.client.grpc.Common.Filter;
import io.qdrant.client.grpc.JsonWithInt.Value;
import io.qdrant.client.grpc.Points.Document;
import io.qdrant.client.grpc.Points.PointStruct;
import io.qdrant.client.grpc.Points.QueryPoints;
import io.qdrant.client.grpc.Points.ScrollPoints;
import java.math.BigInteger;
import java.nio.charset.StandardCharsets;
import java.security.MessageDigest;
import java.time.OffsetDateTime;
import java.time.ZoneOffset;
import java.time.temporal.ChronoUnit;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.UUID;
import java.util.stream.Collectors;
public class Snippet {
// @block-start client-connection
static final String QDRANT_URL = System.getenv("QDRANT_URL");
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
static final QdrantClient client =
new QdrantClient(
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
.withApiKey(QDRANT_API_KEY)
.build());
// @block-end client-connection
// @hide-start
// data and text normalization are not the lesson of this tutorial:
// the full CHUNKS list and normalize() live in the tutorial notebook
static class Chunk {
String url;
String anchor;
int chunkNum;
String text;
String sectionUrl; // derived in prepareChunksForSync
String contentHash; // derived in prepareChunksForSync
String pointId; // derived in prepareChunksForSync
Chunk(String url, String anchor, int chunkNum, String text) {
this.url = url;
this.anchor = anchor;
this.chunkNum = chunkNum;
this.text = text;
}
}
static final List<Chunk> CHUNKS = List.of(
new Chunk(
"https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
"prerequisites",
0,
"Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ..."),
new Chunk(
"https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
"step-3-enable-an-admin-api-key",
0,
"Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ..."));
static String normalize(String text) {
return text.replaceAll("\\s+", " ").strip();
}
// @hide-end
// @block-start create-collection
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
static final String PIPELINE = "docs-prep-pipeline-v1";
static final String COLLECTION = "docs-sync-tutorial";
static void createCollection() throws Exception {
client.createCollectionAsync(
CreateCollection.newBuilder()
.setCollectionName(COLLECTION)
.setVectorsConfig(
VectorsConfig.newBuilder()
.setParams(
VectorParams.newBuilder()
.setSize(384) // all-MiniLM-L6-v2 output dimension
.setDistance(Distance.Cosine)
.build())
.build())
.putAllMetadata(
Map.of(
"embedding_model", value(MODEL),
"pipeline_version", value(PIPELINE)))
.build()).get();
}
// @block-end create-collection
// @block-start check-gate
static void checkGate() throws Exception {
// compare this pipeline's constants against what the collection records about itself
Map<String, Value> meta =
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
Value model = meta.get("embedding_model");
Value pipeline = meta.get("pipeline_version");
if (model == null || !MODEL.equals(model.getStringValue())
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
throw new RuntimeException(
"collection was built by " + meta + ": full re-embed into a fresh collection required");
}
}
// @block-end check-gate
// @block-start identity-and-fingerprint
static String contentHash(String text) throws Exception {
byte[] digest = MessageDigest.getInstance("SHA-256")
.digest(text.getBytes(StandardCharsets.UTF_8));
return String.format("%064x", new BigInteger(1, digest));
}
static String pointId(String url, String anchor, int num) {
// name-based UUID (version 3); the same address always yields the same ID
return UUID.nameUUIDFromBytes(
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
}
// Derive both values (and the section address) for every raw chunk.
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
List<Chunk> out = new ArrayList<>();
for (Chunk c : chunks) {
String text = normalize(c.text);
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
prepared.contentHash = contentHash(text);
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
out.add(prepared);
}
return out;
}
// @block-end identity-and-fingerprint
// @block-start payload
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
Map<String, Value> p = new HashMap<>();
p.put("url", value(chunk.url));
p.put("anchor", value(chunk.anchor));
p.put("chunk_num", value(chunk.chunkNum));
p.put("section_url", value(chunk.sectionUrl));
p.put("text", value(chunk.text));
p.put("content_hash", value(chunk.contentHash));
p.put("last_updated", value(lastUpdated != null
? lastUpdated
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
return p;
}
// @block-end payload
// @block-start payload-indexes
static void createPayloadIndexes() throws Exception {
for (String field : List.of("content_hash", "url", "section_url")) {
client.createPayloadIndexAsync(
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
}
}
// @block-end payload-indexes
// @block-start populate
static void populate() throws Exception {
List<PointStruct> points = new ArrayList<>();
for (Chunk c : prepareChunksForSync(CHUNKS)) {
points.add(
PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build());
}
client.upsertAsync(COLLECTION, points).get();
}
// @block-end populate
// @block-start search
static final String QUERY =
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
static void search() throws Exception {
client.queryAsync(
QueryPoints.newBuilder()
.setCollectionName(COLLECTION)
.setQuery(
nearest(
Document.newBuilder()
.setText(QUERY)
.setModel(MODEL)
.build()))
.setLimit(3)
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
.build()).get();
}
// @block-end search
// @hide-start
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
static List<Chunk> LATEST_CHUNKS;
// @hide-end
// @block-start split-by-state
static class SyncState {
Map<String, Chunk> incoming = new LinkedHashMap<>();
List<Chunk> unchanged = new ArrayList<>();
List<Chunk> contentChanged = new ArrayList<>();
List<Chunk> unknownIds = new ArrayList<>();
}
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
SyncState state = new SyncState();
for (Chunk c : latestChunks) {
state.incoming.put(c.pointId, c);
}
Map<String, String> stored = new HashMap<>();
var points = client.retrieveAsync(
COLLECTION,
state.incoming.keySet().stream()
.map(pid -> id(UUID.fromString(pid)))
.collect(Collectors.toList()),
WithPayloadSelectorFactory.include(List.of("content_hash")),
WithVectorsSelectorFactory.enable(false),
null).get();
for (var p : points) {
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
}
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
String pid = e.getKey();
Chunk c = e.getValue();
if (c.contentHash.equals(stored.get(pid))) {
state.unchanged.add(c);
} else if (stored.containsKey(pid)) {
state.contentChanged.add(c);
} else {
state.unknownIds.add(c);
}
}
return state;
}
// @block-end split-by-state
// @block-start re-embed-changed
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
if (contentChanged.isEmpty()) {
return;
}
List<PointStruct> points = new ArrayList<>();
for (Chunk c : contentChanged) {
points.add(
PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build());
}
client.upsertAsync(COLLECTION, points).get();
}
// @block-end re-embed-changed
// @block-start reuse-or-add
// Reuse an existing embedding when the same text is already stored; embed only what is new.
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
int reused = 0;
int added = 0;
for (Chunk c : unknownIds) {
Filter sameText = Filter.newBuilder()
.addMust(matchKeyword("content_hash", c.contentHash))
.build();
var hits = client.scrollAsync(
ScrollPoints.newBuilder()
.setCollectionName(COLLECTION)
.setFilter(sameText)
.setLimit(1)
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
.setWithVectors(WithVectorsSelectorFactory.enable(true))
.build()).get().getResultList();
PointStruct point;
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
point = PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(vectors(vector(
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
.getDataList())))
.putAllPayload(
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
.build();
reused++;
} else { // genuinely new content: embed and insert
point = PointStruct.newBuilder()
.setId(id(UUID.fromString(c.pointId)))
.setVectors(
vectors(
vector(
Document.newBuilder()
.setText(c.text)
.setModel(MODEL)
.build())))
.putAllPayload(payload(c, null))
.build();
added++;
}
client.upsertAsync(COLLECTION, List.of(point)).get();
}
return new int[] {reused, added};
}
// @block-end reuse-or-add
// @block-start delete-gone
// Remove every point the current crawl no longer contains. Returns how many.
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
if (incomingIds.isEmpty()) {
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
}
Filter stale = Filter.newBuilder()
.addMustNot(hasId(
incomingIds.keySet().stream()
.map(pid -> id(UUID.fromString(pid)))
.collect(Collectors.toList())))
.build();
long toDelete = client.countAsync(COLLECTION, stale, true).get();
// potential check against a threshold to avoid accidental mass deletion could be added here
client.deleteAsync(COLLECTION, stale).get();
return toDelete;
}
// @block-end delete-gone
// @block-start sync
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
checkGate(); // refuse to mix embedding models or pipeline versions
List<Chunk> chunks = prepareChunksForSync(latestChunks);
SyncState state = splitByState(chunks);
reEmbedChanged(state.contentChanged);
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
long deleted = deleteGone(state.incoming);
return Map.of(
"unchanged", (long) state.unchanged.size(),
"re-embedded", (long) state.contentChanged.size(),
"reused_embedding", (long) reusedAdded[0],
"added", (long) reusedAdded[1],
"deleted", deleted);
}
// @block-end sync
// @block-start run-sync
static void runSync() throws Exception {
Map<String, Long> run = sync(LATEST_CHUNKS);
System.out.println(run);
}
// @block-end run-sync
// @hide-start
public static void run() throws Exception {
createCollection();
createPayloadIndexes();
populate();
search();
LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
SyncState state = splitByState(LATEST_CHUNKS);
runSync();
// @hide-end
}
}
@@ -0,0 +1,261 @@
# @block-start client-connection
import os
from qdrant_client import QdrantClient, models
QDRANT_URL = os.getenv("QDRANT_URL")
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
client = QdrantClient(
url=QDRANT_URL,
api_key=QDRANT_API_KEY,
cloud_inference=True
)
# @block-end client-connection
# @hide-start
# data and text normalization are not the lesson of this tutorial:
# the full CHUNKS list and normalize() live in the tutorial notebook
CHUNKS = [
{
"url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
"anchor": "prerequisites",
"chunk_num": 0,
"text": "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
},
{
"url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
"anchor": "step-3-enable-an-admin-api-key",
"chunk_num": 0,
"text": "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
},
]
import re
import unicodedata
def normalize(text):
text = unicodedata.normalize("NFKC", text)
text = text.translate(dict.fromkeys(map(ord, "​‌‍­")))
return re.sub(r"\s+", " ", text).strip()
# @hide-end
# @block-start create-collection
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
PIPELINE = "docs-prep-pipeline-v1"
COLLECTION = "docs-sync-tutorial"
client.create_collection(
COLLECTION,
vectors_config=models.VectorParams(
size=384, # all-MiniLM-L6-v2 output dimension
distance=models.Distance.COSINE,
),
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
)
# @block-end create-collection
# @block-start check-gate
def check_gate():
# compare this pipeline's constants against what the collection records about itself
meta = client.get_collection(COLLECTION).config.metadata or {}
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
# @block-end check-gate
# @block-start identity-and-fingerprint
import hashlib
import uuid
from datetime import datetime, timezone
def content_hash(text):
return hashlib.sha256(text.encode()).hexdigest()
def point_id(url, anchor, num):
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
def prepare_chunks_for_sync(chunks):
"""Derive both values (and the section address) for every raw chunk."""
out = []
for c in chunks:
text = normalize(c["text"])
out.append({
**c,
"text": text,
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
"content_hash": content_hash(text),
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
})
return out
# @block-end identity-and-fingerprint
# @block-start payload
def payload(chunk, last_updated=None):
return {
"url": chunk["url"],
"anchor": chunk["anchor"],
"chunk_num": chunk["chunk_num"],
"section_url": chunk["section_url"],
"text": chunk["text"],
"content_hash": chunk["content_hash"],
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
}
# @block-end payload
# @block-start payload-indexes
for field in ("content_hash", "url", "section_url"):
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
# @block-end payload-indexes
# @block-start populate
client.upsert(COLLECTION, points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
for c in prepare_chunks_for_sync(CHUNKS)
], wait=True)
# @block-end populate
# @block-start search
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
client.query_points(
COLLECTION,
query=models.Document(text=QUERY, model=MODEL),
limit=3,
with_payload=["section_url", "text"],
)
# @block-end search
# @hide-start
# the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
LATEST_CHUNKS = prepare_chunks_for_sync(CHUNKS)
# @hide-end
# @block-start split-by-state
def split_by_state(latest_chunks):
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
incoming = {c["point_id"]: c for c in latest_chunks}
stored = {}
points = client.retrieve(
COLLECTION,
ids=list(incoming),
with_payload=["content_hash"],
with_vectors=False,
)
for p in points:
stored[str(p.id)] = p.payload["content_hash"]
unchanged, content_changed, unknown_ids = [], [], []
for pid, c in incoming.items():
if stored.get(pid) == c["content_hash"]:
unchanged.append(c)
elif pid in stored:
content_changed.append(c)
else:
unknown_ids.append(c)
return incoming, unchanged, content_changed, unknown_ids
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
# @block-end split-by-state
# @block-start re-embed-changed
def re_embed_changed(content_changed):
if not content_changed:
return
client.upsert(COLLECTION,
points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
for c in content_changed],
wait=True)
# @block-end re-embed-changed
# @block-start reuse-or-add
def reuse_or_add(unknown_ids):
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
reused, added = 0, 0
for c in unknown_ids:
same_text = models.Filter(must=[
models.FieldCondition(
key="content_hash",
match=models.MatchValue(value=c["content_hash"]),
)
])
hits, _ = client.scroll(
COLLECTION,
scroll_filter=same_text,
limit=1,
with_payload=["last_updated"],
with_vectors=True,
)
if hits: # same text, new address: copy the vector, keep its last_updated
point = models.PointStruct(
id=c["point_id"],
vector=hits[0].vector,
payload=payload(c, hits[0].payload["last_updated"]),
)
reused += 1
else: # genuinely new content: embed and insert
point = models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
added += 1
client.upsert(COLLECTION, points=[point], wait=True)
return reused, added
# @block-end reuse-or-add
# @block-start delete-gone
def delete_gone(incoming_ids):
"""Remove every point the current crawl no longer contains. Returns how many."""
if not incoming_ids:
raise ValueError("Refusing to delete from an empty source snapshot.")
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
to_delete = client.count(COLLECTION, count_filter=stale).count
# potential check against a threshold to avoid accidental mass deletion could be added here
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
return to_delete
# @block-end delete-gone
# @block-start sync
def sync(latest_chunks):
check_gate() # refuse to mix embedding models or pipeline versions
chunks = prepare_chunks_for_sync(latest_chunks)
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
re_embed_changed(content_changed)
reused, added = reuse_or_add(unknown_ids)
deleted = delete_gone(incoming_ids)
return {
"unchanged": len(unchanged),
"re-embedded": len(content_changed),
"reused_embedding": reused,
"added": added,
"deleted": deleted,
}
# @block-end sync
# @block-start run-sync
run = sync(LATEST_CHUNKS)
print(run)
# @block-end run-sync
@@ -0,0 +1,397 @@
use serde_json::{json, Value};
use std::collections::HashMap;
use qdrant_client::qdrant::{
point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
};
use qdrant_client::{Payload, Qdrant};
use sha2::{Digest, Sha256};
pub async fn main() -> anyhow::Result<()> {
// @block-start client-connection
let qdrant_url = std::env::var("QDRANT_URL")?;
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
let client = Qdrant::from_url(&qdrant_url)
.api_key(qdrant_api_key)
.build()?;
// @block-end client-connection
// @hide-start
// data and text normalization are not the lesson of this tutorial:
// the full CHUNKS list and normalize() live in the tutorial notebook
#[derive(Clone, Default)]
struct Chunk {
url: String,
anchor: String,
chunk_num: u32,
text: String,
section_url: String,
content_hash: String,
point_id: String,
}
let chunks: Vec<Chunk> = vec![
Chunk {
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
anchor: "prerequisites".into(),
chunk_num: 0,
text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...".into(),
..Default::default()
},
Chunk {
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
anchor: "step-3-enable-an-admin-api-key".into(),
chunk_num: 0,
text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...".into(),
..Default::default()
},
];
fn normalize(text: &str) -> String {
text.split_whitespace().collect::<Vec<_>>().join(" ")
}
// @hide-end
// @block-start create-collection
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
const PIPELINE: &str = "docs-prep-pipeline-v1";
const COLLECTION: &str = "docs-sync-tutorial";
let mut metadata: HashMap<String, Value> = HashMap::new();
metadata.insert("embedding_model".to_string(), json!(MODEL));
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
client
.create_collection(
CreateCollectionBuilder::new(COLLECTION)
.vectors_config(VectorParamsBuilder::new(
384, // all-MiniLM-L6-v2 output dimension
Distance::Cosine,
))
.metadata(metadata),
)
.await?;
// @block-end create-collection
// @block-start check-gate
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
// compare this pipeline's constants against what the collection records about itself
let meta = client
.collection_info(COLLECTION)
.await?
.result
.and_then(|info| info.config)
.map(|config| config.metadata)
.unwrap_or_default();
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
!= Some(PIPELINE)
{
anyhow::bail!(
"collection was built by {meta:?}: full re-embed into a fresh collection required"
);
}
Ok(())
}
// @block-end check-gate
// @block-start identity-and-fingerprint
fn content_hash(text: &str) -> String {
Sha256::digest(text.as_bytes())
.iter()
.map(|byte| format!("{byte:02x}"))
.collect()
}
fn point_id(url: &str, anchor: &str, num: u32) -> String {
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
uuid::Uuid::new_v5(
&uuid::Uuid::NAMESPACE_URL,
format!("{url}#{anchor}::{num}").as_bytes(),
)
.to_string()
}
/// Derive both values (and the section address) for every raw chunk.
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
chunks
.iter()
.map(|c| {
let text = normalize(&c.text);
Chunk {
text: text.clone(),
section_url: if c.anchor.is_empty() {
c.url.clone()
} else {
format!("{}#{}", c.url, c.anchor)
},
content_hash: content_hash(&text),
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
..c.clone()
}
})
.collect()
}
// @block-end identity-and-fingerprint
// @block-start payload
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
let last_updated = last_updated.unwrap_or_else(|| {
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
});
Ok(Payload::try_from(serde_json::json!({
"url": chunk.url,
"anchor": chunk.anchor,
"chunk_num": chunk.chunk_num,
"section_url": chunk.section_url,
"text": chunk.text,
"content_hash": chunk.content_hash,
"last_updated": last_updated,
}))?)
}
// @block-end payload
// @block-start payload-indexes
for field in ["content_hash", "url", "section_url"] {
client
.create_field_index(CreateFieldIndexCollectionBuilder::new(
COLLECTION,
field,
FieldType::Keyword,
))
.await?;
}
// @block-end payload-indexes
// @block-start populate
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
.iter()
.map(|c| {
Ok(PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
))
})
.collect::<anyhow::Result<_>>()?;
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
.await?;
// @block-end populate
// @block-start search
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
client
.query(
QueryPointsBuilder::new(COLLECTION)
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
.limit(3)
.with_payload(PayloadIncludeSelector::new(vec![
"section_url".to_string(),
"text".to_string(),
])),
)
.await?;
// @block-end search
// @hide-start
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
let latest_chunks = prepare_chunks_for_sync(&chunks);
// @hide-end
// @block-start split-by-state
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async fn split_by_state(
client: &Qdrant,
latest_chunks: &[Chunk],
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
let incoming: HashMap<String, Chunk> = latest_chunks
.iter()
.map(|c| (c.point_id.clone(), c.clone()))
.collect();
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
let points = client
.get_points(
GetPointsBuilder::new(COLLECTION, ids)
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
.with_vectors(false),
)
.await?;
let mut stored: HashMap<String, String> = HashMap::new();
for p in points.result {
let hash = p.get("content_hash").as_str().cloned();
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
(p.id.and_then(|i| i.point_id_options), hash)
{
stored.insert(id, hash);
}
}
let (mut unchanged, mut content_changed, mut unknown_ids) =
(Vec::new(), Vec::new(), Vec::new());
for (pid, c) in &incoming {
if stored.get(pid) == Some(&c.content_hash) {
unchanged.push(c.clone());
} else if stored.contains_key(pid) {
content_changed.push(c.clone());
} else {
unknown_ids.push(c.clone());
}
}
Ok((incoming, unchanged, content_changed, unknown_ids))
}
let (incoming_ids, unchanged, content_changed, unknown_ids) =
split_by_state(&client, &latest_chunks).await?;
// @block-end split-by-state
// @hide-start
_ = (&incoming_ids, &unchanged, &content_changed, &unknown_ids);
// @hide-end
// @block-start re-embed-changed
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
if content_changed.is_empty() {
return Ok(());
}
let points: Vec<PointStruct> = content_changed
.iter()
.map(|c| {
Ok(PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
))
})
.collect::<anyhow::Result<_>>()?;
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
.await?;
Ok(())
}
// @block-end re-embed-changed
// @block-start reuse-or-add
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
let (mut reused, mut added) = (0, 0);
for c in unknown_ids {
let same_text =
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
let hits = client
.scroll(
ScrollPointsBuilder::new(COLLECTION)
.filter(same_text)
.limit(1)
.with_payload(PayloadIncludeSelector::new(vec![
"last_updated".to_string()
]))
.with_vectors(true),
)
.await?
.result;
let point = if let Some(hit) = hits.into_iter().next() {
// same text, new address: copy the vector, keep its last_updated
let last_updated = hit.get("last_updated").as_str().cloned();
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
Some(vector_output::Vector::Dense(dense)) => dense.data,
_ => anyhow::bail!("expected a dense vector on the stored point"),
},
_ => anyhow::bail!("expected a dense vector on the stored point"),
};
reused += 1;
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
} else {
// genuinely new content: embed and insert
added += 1;
PointStruct::new(
c.point_id.clone(),
Document::new(&c.text, MODEL),
payload(c, None)?,
)
};
client
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
.await?;
}
Ok((reused, added))
}
// @block-end reuse-or-add
// @block-start delete-gone
/// Remove every point the current crawl no longer contains. Returns how many.
async fn delete_gone(
client: &Qdrant,
incoming_ids: &HashMap<String, Chunk>,
) -> anyhow::Result<u64> {
if incoming_ids.is_empty() {
anyhow::bail!("Refusing to delete from an empty source snapshot.");
}
let stale = Filter::must_not([Condition::has_id(
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
)]);
let to_delete = client
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
.await?
.result
.map(|r| r.count)
.unwrap_or(0);
// potential check against a threshold to avoid accidental mass deletion could be added here
client
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
.await?;
Ok(to_delete)
}
// @block-end delete-gone
// @block-start sync
async fn sync(
client: &Qdrant,
latest_chunks: &[Chunk],
) -> anyhow::Result<HashMap<&'static str, usize>> {
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
let chunks = prepare_chunks_for_sync(latest_chunks);
let (incoming_ids, unchanged, content_changed, unknown_ids) =
split_by_state(client, &chunks).await?;
re_embed_changed(client, &content_changed).await?;
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
let deleted = delete_gone(client, &incoming_ids).await?;
Ok(HashMap::from([
("unchanged", unchanged.len()),
("re-embedded", content_changed.len()),
("reused_embedding", reused),
("added", added),
("deleted", deleted as usize),
]))
}
// @block-end sync
// @block-start run-sync
let run = sync(&client, &latest_chunks).await?;
println!("{run:?}");
// @block-end run-sync
Ok(())
}
@@ -0,0 +1,288 @@
// @block-start client-connection
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
const QDRANT_URL = process.env.QDRANT_URL;
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
const client = new QdrantClient({
url: QDRANT_URL,
apiKey: QDRANT_API_KEY,
});
// @block-end client-connection
// @hide-start
// data and text normalization are not the lesson of this tutorial:
// the full CHUNKS list and normalize() live in the tutorial notebook
const CHUNKS = [
{
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
anchor: "prerequisites",
chunk_num: 0,
text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
},
{
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
anchor: "step-3-enable-an-admin-api-key",
chunk_num: 0,
text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
},
];
function normalize(text: string): string {
return text
.normalize("NFKC")
.replace(/[\u200B\u200C\u200D\uFEFF\u00AD]/g, "")
.replace(/\s+/g, " ")
.trim();
}
// @hide-end
// @block-start create-collection
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
const PIPELINE = "docs-prep-pipeline-v1";
const COLLECTION = "docs-sync-tutorial";
await client.createCollection(COLLECTION, {
vectors: {
size: 384, // all-MiniLM-L6-v2 output dimension
distance: "Cosine",
},
});
await client.updateCollection(COLLECTION, {
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
});
// @block-end create-collection
// @block-start check-gate
async function checkGate() {
// compare this pipeline's constants against what the collection records about itself
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
{}) as Record<string, unknown>;
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
}
}
// @block-end check-gate
// @block-start identity-and-fingerprint
import { createHash } from "node:crypto";
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
function contentHash(text: string): string {
return createHash("sha256").update(text).digest("hex");
}
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
function pointId(url: string, anchor: string, num: number): string {
// Qdrant accepts any well-formed UUID as a point ID:
// hash the address, format the digest as a UUID, and the same address always yields the same ID
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
}
// Derive both values (and the section address) for every raw chunk.
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
return chunks.map((c) => {
const text = normalize(c.text);
return {
...c,
text,
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
content_hash: contentHash(text),
point_id: pointId(c.url, c.anchor, c.chunk_num),
};
});
}
// @block-end identity-and-fingerprint
// @block-start payload
function payload(chunk: SyncChunk, lastUpdated?: string) {
return {
url: chunk.url,
anchor: chunk.anchor,
chunk_num: chunk.chunk_num,
section_url: chunk.section_url,
text: chunk.text,
content_hash: chunk.content_hash,
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
};
}
// @block-end payload
// @block-start payload-indexes
for (const field of ["content_hash", "url", "section_url"]) {
await client.createPayloadIndex(COLLECTION, {
field_name: field,
field_schema: "keyword",
});
}
// @block-end payload-indexes
// @block-start populate
await client.upsert(COLLECTION, {
points: prepareChunksForSync(CHUNKS).map((c) => ({
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
})),
wait: true,
});
// @block-end populate
// @block-start search
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
await client.query(COLLECTION, {
query: { text: QUERY, model: MODEL },
limit: 3,
with_payload: ["section_url", "text"],
});
// @block-end search
// @hide-start
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
const LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
// @hide-end
// @block-start split-by-state
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
async function splitByState(latestChunks: SyncChunk[]) {
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
const stored = new Map<string, string>();
const points = await client.retrieve(COLLECTION, {
ids: [...incoming.keys()],
with_payload: ["content_hash"],
with_vector: false,
});
for (const p of points) {
stored.set(String(p.id), p.payload?.content_hash as string);
}
const unchanged: SyncChunk[] = [];
const contentChanged: SyncChunk[] = [];
const unknownIds: SyncChunk[] = [];
for (const [pid, c] of incoming) {
if (stored.get(pid) === c.content_hash) {
unchanged.push(c);
} else if (stored.has(pid)) {
contentChanged.push(c);
} else {
unknownIds.push(c);
}
}
return { incoming, unchanged, contentChanged, unknownIds };
}
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
// @block-end split-by-state
// @block-start re-embed-changed
async function reEmbedChanged(contentChanged: SyncChunk[]) {
if (contentChanged.length === 0) {
return;
}
await client.upsert(COLLECTION, {
points: contentChanged.map((c) => ({
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
})),
wait: true,
});
}
// @block-end re-embed-changed
// @block-start reuse-or-add
// Reuse an existing embedding when the same text is already stored; embed only what is new.
async function reuseOrAdd(unknownIds: SyncChunk[]) {
let reused = 0;
let added = 0;
for (const c of unknownIds) {
const sameText = {
must: [
{
key: "content_hash",
match: { value: c.content_hash },
},
],
};
const hits = (await client.scroll(COLLECTION, {
filter: sameText,
limit: 1,
with_payload: ["last_updated"],
with_vector: true,
})).points;
let point: Schemas["PointStruct"];
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
point = {
id: c.point_id,
vector: hits[0].vector as number[],
payload: payload(c, hits[0].payload?.last_updated as string),
};
reused += 1;
} else { // genuinely new content: embed and insert
point = {
id: c.point_id,
vector: { text: c.text, model: MODEL },
payload: payload(c),
};
added += 1;
}
await client.upsert(COLLECTION, { points: [point], wait: true });
}
return { reused, added };
}
// @block-end reuse-or-add
// @block-start delete-gone
// Remove every point the current crawl no longer contains. Returns how many.
async function deleteGone(incoming: Map<string, SyncChunk>) {
if (incoming.size === 0) {
throw new Error("Refusing to delete from an empty source snapshot.");
}
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
// potential check against a threshold to avoid accidental mass deletion could be added here
await client.delete(COLLECTION, { filter: stale, wait: true });
return toDelete;
}
// @block-end delete-gone
// @block-start sync
async function sync(latestChunks: RawChunk[]) {
await checkGate(); // refuse to mix embedding models or pipeline versions
const chunks = prepareChunksForSync(latestChunks);
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
await reEmbedChanged(contentChanged);
const { reused, added } = await reuseOrAdd(unknownIds);
const deleted = await deleteGone(incoming);
return {
"unchanged": unchanged.length,
"re-embedded": contentChanged.length,
"reused_embedding": reused,
"added": added,
"deleted": deleted,
};
}
// @block-end sync
// @block-start run-sync
const run = await sync(LATEST_CHUNKS);
console.log(run);
// @block-end run-sync
@@ -35,27 +35,13 @@ The tutorial has an accompanying [notebook](https://github.com/qdrant/examples/b
## Prerequisites
```python
%pip install -q "qdrant-client>=1.18"
```
Install the [Qdrant client of your choice](/documentation/interfaces/#client-libraries).
We use Qdrant Cloud and its [Free Embedding Inference](/documentation/cloud/inference/#free-embedding-models).
Create a Free Tier [Qdrant Cloud cluster](https://cloud.qdrant.io/) and set `QDRANT_URL` and `QDRANT_API_KEY` in your environment.
```python
import os
from qdrant_client import QdrantClient, models
QDRANT_URL = os.getenv("QDRANT_URL")
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
client = QdrantClient(
url=QDRANT_URL,
api_key=QDRANT_API_KEY,
cloud_inference=True
)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="client-connection" >}}
## The Data: Qdrant Documentation
@@ -133,31 +119,11 @@ Vectors produced by different embedding models, or by the same model over differ
Let's consider a simple guardrail: save which model and which pipeline version produced the data points, in [**collection metadata**](/documentation/manage-data/collections/#collection-metadata), and verify against it. If one of the two changed, we need to trigger full collection re-embedding.
```python
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
PIPELINE = "docs-prep-pipeline-v1"
COLLECTION = "docs-sync-tutorial"
client.create_collection(
COLLECTION,
vectors_config=models.VectorParams(
size=384, # all-MiniLM-L6-v2 output dimension
distance=models.Distance.COSINE,
),
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="create-collection" >}}
The gate against mixing embedding generations is then a simple check at the start of every run:
```python
def check_gate():
# compare this pipeline's constants against what the collection records about itself
meta = client.get_collection(COLLECTION).config.metadata or {}
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="check-gate" >}}
## Characteristics of a Document Chunk
@@ -174,32 +140,7 @@ Hence every record should get two derived values:
- **Content fingerprint**, like SHA-256 of the text. It changes if a single character changes, and never otherwise. Comparing fingerprints answers "*Is it the same content?*" without comparing texts.
- **Deterministic ID** for position in documentation. For example, `url + "#" + anchor + "::" + chunk_num` turned into a UUID, one of the two point ID formats Qdrant accepts. Comparing IDs answers "*Is this content still at the same position?*".
```python
import hashlib
import uuid
from datetime import datetime, timezone
def content_hash(text):
return hashlib.sha256(text.encode()).hexdigest()
def point_id(url, anchor, num):
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
def prepare_chunks_for_sync(chunks):
"""Derive both values (and the section address) for every raw chunk."""
out = []
for c in chunks:
text = normalize(c["text"])
out.append({
**c,
"text": text,
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
"content_hash": content_hash(text),
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
})
return out
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="identity-and-fingerprint" >}}
Example:
```text
@@ -218,56 +159,24 @@ Additionally, a point can be described by the following fields:
<details>
<summary>payload() implementation</summary>
```python
def payload(chunk, last_updated=None):
return {
"url": chunk["url"],
"anchor": chunk["anchor"],
"chunk_num": chunk["chunk_num"],
"section_url": chunk["section_url"],
"text": chunk["text"],
"content_hash": chunk["content_hash"],
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
}
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload" >}}
</details>
For all the payload fields used for filtering or grouping we need to create a [**payload index**](/documentation/manage-data/indexing/).
```python
for field in ("content_hash", "url", "section_url"):
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload-indexes" >}}
## Populate Collection
Populate the collection with the whole documentation.
```python
client.upsert(COLLECTION, points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL), # Cloud Inference embeds text server-side
payload=payload(c),
)
for c in prepare_chunks_for_sync(CHUNKS)
], wait=True)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="populate" >}}
<details>
<summary>Test the search against it</summary>
```python
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
client.query_points(
COLLECTION,
query=models.Document(text=QUERY, model=MODEL),
limit=3,
with_payload=["section_url", "text"],
)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="search" >}}
You should get something like:
@@ -354,35 +263,7 @@ We now check every incoming chunk against the collection: does its ID (address)
[`retrieve`](/documentation/manage-data/points/) fetches points by ID. At corpus scale you would batch the IDs.
```python
def split_by_state(latest_chunks):
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
incoming = {c["point_id"]: c for c in latest_chunks}
stored = {}
points = client.retrieve(
COLLECTION,
ids=list(incoming),
with_payload=["content_hash"],
with_vectors=False,
)
for p in points:
stored[str(p.id)] = p.payload["content_hash"]
unchanged, content_changed, unknown_ids = [], [], []
for pid, c in incoming.items():
if stored.get(pid) == c["content_hash"]:
unchanged.append(c)
elif pid in stored:
content_changed.append(c)
else:
unknown_ids.append(c)
return incoming, unchanged, content_changed, unknown_ids
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="split-by-state" >}}
### Case 1: Unchanged, Do Nothing
@@ -393,20 +274,7 @@ These chunks carry the same fingerprint as before.
The chunk about Step 3 exists under a known ID (it didn't change its position on the docs website) but carries new information.
Use `upsert`: writing a point under an existing ID replaces it.
```python
def re_embed_changed(content_changed):
if not content_changed:
return
client.upsert(COLLECTION,
points=[
models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
for c in content_changed],
wait=True)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="re-embed-changed" >}}
### Cases 3 and 4: ID Is Not Present in the Collection
@@ -418,45 +286,7 @@ A filtered [`scroll`](/documentation/manage-data/points/) on `content_hash` answ
**Note:** *This version performs one hash lookup per unknown chunk so the decision is easy to inspect. In production, batch hash lookups and point upserts.*
```python
def reuse_or_add(unknown_ids):
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
reused, added = 0, 0
for c in unknown_ids:
same_text = models.Filter(must=[
models.FieldCondition(
key="content_hash",
match=models.MatchValue(value=c["content_hash"]),
)
])
hits, _ = client.scroll(
COLLECTION,
scroll_filter=same_text,
limit=1,
with_payload=["last_updated"],
with_vectors=True,
)
if hits: # same text, new address: copy the vector, keep its last_updated
point = models.PointStruct(
id=c["point_id"],
vector=hits[0].vector,
payload=payload(c, hits[0].payload["last_updated"]),
)
reused += 1
else: # genuinely new content: embed and insert
point = models.PointStruct(
id=c["point_id"],
vector=models.Document(text=c["text"], model=MODEL),
payload=payload(c),
)
added += 1
client.upsert(COLLECTION, points=[point], wait=True)
return reused, added
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="reuse-or-add" >}}
What's important to notice: the old points, the migration page under its old URL, are still in the collection. They need to be removed, and that is the last case.
@@ -471,51 +301,17 @@ Whatever LATEST_CHUNKS does not contain no longer exists at the source. The dele
**Note:** Frequent re-embeddings and deletions don't degrade the index over time: background [optimizers](/documentation/ops-optimization/optimizer/) rebuild and merge index segments as changes accumulate.
```python
def delete_gone(incoming_ids):
"""Remove every point the current crawl no longer contains. Returns how many."""
if not incoming_ids:
raise ValueError("Refusing to delete from an empty source snapshot.")
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
to_delete = client.count(COLLECTION, count_filter=stale).count
# potential check against a threshold to avoid accidental mass deletion could be added here
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
return to_delete
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="delete-gone" >}}
## Run and Verify the Sync
The five cases, assembled from the functions defined above:
```python
def sync(latest_chunks):
check_gate() # refuse to mix embedding models or pipeline versions
chunks = prepare_chunks_for_sync(latest_chunks)
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
re_embed_changed(content_changed)
reused, added = reuse_or_add(unknown_ids)
deleted = delete_gone(incoming_ids)
return {
"unchanged": len(unchanged),
"re-embedded": len(content_changed),
"reused_embedding": reused,
"added": added,
"deleted": deleted,
}
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="sync" >}}
Run the sync.
```python
run = sync(LATEST_CHUNKS)
print(run)
```
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="run-sync" >}}
You should see something like: