mirror of
https://github.com/qdrant/landing_page.git
synced 2026-10-06 19:38:30 +02:00
added snippets in all languages
This commit is contained in:
@@ -15,4 +15,5 @@ serde_json = "1.0.145"
|
||||
tempfile = "3"
|
||||
tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] }
|
||||
ureq = { version = "3", features = ["json"] }
|
||||
uuid = { version = "1.18.1", features = ["v4"] }
|
||||
uuid = { version = "1.18.1", features = ["v4", "v5"] }
|
||||
sha2 = "0.11"
|
||||
|
||||
@@ -6,6 +6,6 @@
|
||||
| [Time-Based Sharding](/documentation/tutorials-operations/time-based-sharding/) | Efficiently manage time-series data with user-defined sharding. | <span class="pill">Any</span> | 1h | <span class="text-yellow">Intermediate</span> |
|
||||
| [Large-Scale Search](/documentation/tutorials-operations/large-scale-search/) | Cost-efficient search for LAION-400M datasets. | <span class="pill">Any</span> | 48h | <span class="text-red">Advanced</span> |
|
||||
| [Secure a Self-Hosted Instance](/documentation/tutorials-operations/secure-qdrant/) | Enable TLS, API keys, and JWT access control. | <span class="pill">Any</span> | 45m | <span class="text-yellow">Intermediate</span> |
|
||||
| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | <span class="pill">Python</span> | 25m | <span class="text-green">Beginner</span> |
|
||||
| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | <span class="pill">Any</span> | 25m | <span class="text-green">Beginner</span> |
|
||||
| [Qdrant Cloud Prometheus Monitoring](/documentation/ops-monitoring/managed-cloud-prometheus/) | Observability with Prometheus and Grafana. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
|
||||
| [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
|
||||
+310
@@ -0,0 +1,310 @@
|
||||
using System.Security.Cryptography;
|
||||
using System.Text;
|
||||
using System.Text.RegularExpressions;
|
||||
using Qdrant.Client;
|
||||
using Qdrant.Client.Grpc;
|
||||
using static Qdrant.Client.Grpc.Conditions;
|
||||
using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
|
||||
|
||||
public class Snippet
|
||||
{
|
||||
public static async Task Run()
|
||||
{
|
||||
// @block-start client-connection
|
||||
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
|
||||
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
|
||||
|
||||
var client = new QdrantClient(
|
||||
host: QDRANT_URL!,
|
||||
https: true,
|
||||
apiKey: QDRANT_API_KEY
|
||||
);
|
||||
// @block-end client-connection
|
||||
|
||||
// @hide-start
|
||||
// data and text normalization are not the lesson of this tutorial:
|
||||
// the full CHUNKS list and Normalize() live in the tutorial notebook
|
||||
var CHUNKS = new List<Chunk>
|
||||
{
|
||||
(
|
||||
Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
Anchor: "prerequisites",
|
||||
ChunkNum: 0,
|
||||
Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||
SectionUrl: "", ContentHash: "", PointId: ""
|
||||
),
|
||||
(
|
||||
Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
Anchor: "step-3-enable-an-admin-api-key",
|
||||
ChunkNum: 0,
|
||||
Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||
SectionUrl: "", ContentHash: "", PointId: ""
|
||||
),
|
||||
};
|
||||
|
||||
string Normalize(string text) => Regex.Replace(text, @"\s+", " ").Trim();
|
||||
// @hide-end
|
||||
|
||||
// @block-start create-collection
|
||||
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
var PIPELINE = "docs-prep-pipeline-v1";
|
||||
var COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
await client.CreateCollectionAsync(
|
||||
collectionName: COLLECTION,
|
||||
vectorsConfig: new VectorParams
|
||||
{
|
||||
Size = 384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance = Distance.Cosine
|
||||
},
|
||||
metadata: new()
|
||||
{
|
||||
["embedding_model"] = MODEL,
|
||||
["pipeline_version"] = PIPELINE
|
||||
}
|
||||
);
|
||||
// @block-end create-collection
|
||||
|
||||
// @block-start check-gate
|
||||
async Task CheckGate()
|
||||
{
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
|
||||
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
|
||||
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
|
||||
|
||||
if (model != MODEL || pipeline != PIPELINE)
|
||||
throw new InvalidOperationException(
|
||||
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
|
||||
}
|
||||
// @block-end check-gate
|
||||
|
||||
// @block-start identity-and-fingerprint
|
||||
string ContentHash(string text) =>
|
||||
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
|
||||
|
||||
// Qdrant accepts any well-formed UUID as a point ID:
|
||||
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
|
||||
string PointIdFor(string url, string anchor, int num) =>
|
||||
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
|
||||
{
|
||||
var prepared = new List<Chunk>();
|
||||
foreach (var c in chunks)
|
||||
{
|
||||
var text = Normalize(c.Text);
|
||||
prepared.Add(c with
|
||||
{
|
||||
Text = text,
|
||||
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
|
||||
ContentHash = ContentHash(text),
|
||||
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
|
||||
});
|
||||
}
|
||||
return prepared;
|
||||
}
|
||||
// @block-end identity-and-fingerprint
|
||||
|
||||
// @block-start payload
|
||||
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
|
||||
{
|
||||
["url"] = chunk.Url,
|
||||
["anchor"] = chunk.Anchor,
|
||||
["chunk_num"] = chunk.ChunkNum,
|
||||
["section_url"] = chunk.SectionUrl,
|
||||
["text"] = chunk.Text,
|
||||
["content_hash"] = chunk.ContentHash,
|
||||
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
|
||||
};
|
||||
// @block-end payload
|
||||
|
||||
// @block-start payload-indexes
|
||||
foreach (var field in new[] { "content_hash", "url", "section_url" })
|
||||
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
|
||||
// @block-end payload-indexes
|
||||
|
||||
// @block-start populate
|
||||
await client.UpsertAsync(
|
||||
collectionName: COLLECTION,
|
||||
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
}).ToList(),
|
||||
wait: true
|
||||
);
|
||||
// @block-end populate
|
||||
|
||||
// @block-start search
|
||||
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
await client.QueryAsync(
|
||||
collectionName: COLLECTION,
|
||||
query: new Document { Text = QUERY, Model = MODEL },
|
||||
limit: 3,
|
||||
payloadSelector: new[] { "section_url", "text" }
|
||||
);
|
||||
// @block-end search
|
||||
|
||||
// @hide-start
|
||||
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||
var LATEST_CHUNKS = PrepareChunksForSync(CHUNKS);
|
||||
// @hide-end
|
||||
|
||||
// @block-start split-by-state
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
|
||||
SplitByState(List<Chunk> latestChunks)
|
||||
{
|
||||
var incoming = latestChunks.ToDictionary(c => c.PointId);
|
||||
|
||||
var stored = new Dictionary<string, string>();
|
||||
var points = await client.RetrieveAsync(
|
||||
COLLECTION,
|
||||
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
|
||||
payloadSelector: new[] { "content_hash" },
|
||||
vectorSelector: false
|
||||
);
|
||||
foreach (var p in points)
|
||||
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
|
||||
|
||||
var unchanged = new List<Chunk>();
|
||||
var contentChanged = new List<Chunk>();
|
||||
var unknownIds = new List<Chunk>();
|
||||
foreach (var (pid, c) in incoming)
|
||||
{
|
||||
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
|
||||
unchanged.Add(c);
|
||||
else if (stored.ContainsKey(pid))
|
||||
contentChanged.Add(c);
|
||||
else
|
||||
unknownIds.Add(c);
|
||||
}
|
||||
|
||||
return (incoming, unchanged, contentChanged, unknownIds);
|
||||
}
|
||||
|
||||
var splitState = await SplitByState(LATEST_CHUNKS);
|
||||
// @block-end split-by-state
|
||||
|
||||
// @block-start re-embed-changed
|
||||
async Task ReEmbedChanged(List<Chunk> contentChanged)
|
||||
{
|
||||
if (contentChanged.Count == 0)
|
||||
return;
|
||||
await client.UpsertAsync(
|
||||
collectionName: COLLECTION,
|
||||
points: contentChanged.Select(c => new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
}).ToList(),
|
||||
wait: true
|
||||
);
|
||||
}
|
||||
// @block-end re-embed-changed
|
||||
|
||||
// @block-start reuse-or-add
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
|
||||
{
|
||||
int reused = 0, added = 0;
|
||||
|
||||
foreach (var c in unknownIds)
|
||||
{
|
||||
var sameText = new Filter
|
||||
{
|
||||
Must = { MatchKeyword("content_hash", c.ContentHash) }
|
||||
};
|
||||
var hits = (await client.ScrollAsync(
|
||||
COLLECTION,
|
||||
filter: sameText,
|
||||
limit: 1,
|
||||
payloadSelector: new[] { "last_updated" },
|
||||
vectorsSelector: true
|
||||
)).Result;
|
||||
|
||||
PointStruct point;
|
||||
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
|
||||
{
|
||||
point = new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
|
||||
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
|
||||
};
|
||||
reused++;
|
||||
}
|
||||
else // genuinely new content: embed and insert
|
||||
{
|
||||
point = new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
};
|
||||
added++;
|
||||
}
|
||||
|
||||
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
|
||||
}
|
||||
|
||||
return (reused, added);
|
||||
}
|
||||
// @block-end reuse-or-add
|
||||
|
||||
// @block-start delete-gone
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
|
||||
{
|
||||
if (incomingIds.Count == 0)
|
||||
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
|
||||
|
||||
var stale = new Filter
|
||||
{
|
||||
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
|
||||
};
|
||||
|
||||
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
|
||||
return toDelete;
|
||||
}
|
||||
// @block-end delete-gone
|
||||
|
||||
// @block-start sync
|
||||
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
|
||||
{
|
||||
await CheckGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
var chunks = PrepareChunksForSync(latestChunks);
|
||||
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
|
||||
|
||||
await ReEmbedChanged(contentChanged);
|
||||
var (reused, added) = await ReuseOrAdd(unknownIds);
|
||||
var deleted = await DeleteGone(incomingIds);
|
||||
|
||||
return new Dictionary<string, long>
|
||||
{
|
||||
["unchanged"] = unchanged.Count,
|
||||
["re-embedded"] = contentChanged.Count,
|
||||
["reused_embedding"] = reused,
|
||||
["added"] = added,
|
||||
["deleted"] = (long)deleted,
|
||||
};
|
||||
}
|
||||
// @block-end sync
|
||||
|
||||
// @block-start run-sync
|
||||
var run = await Sync(LATEST_CHUNKS);
|
||||
foreach (var (op, count) in run)
|
||||
Console.WriteLine($"{op}: {count}");
|
||||
// @block-end run-sync
|
||||
}
|
||||
|
||||
}
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
```csharp
|
||||
async Task CheckGate()
|
||||
{
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
|
||||
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
|
||||
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
|
||||
|
||||
if (model != MODEL || pipeline != PIPELINE)
|
||||
throw new InvalidOperationException(
|
||||
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
|
||||
}
|
||||
```
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
```go
|
||||
checkGate := func() {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
|
||||
meta := info.GetConfig().GetMetadata()
|
||||
|
||||
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
|
||||
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
|
||||
}
|
||||
}
|
||||
```
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
```java
|
||||
static void checkGate() throws Exception {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
Map<String, Value> meta =
|
||||
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
|
||||
|
||||
Value model = meta.get("embedding_model");
|
||||
Value pipeline = meta.get("pipeline_version");
|
||||
if (model == null || !MODEL.equals(model.getStringValue())
|
||||
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
|
||||
throw new RuntimeException(
|
||||
"collection was built by " + meta + ": full re-embed into a fresh collection required");
|
||||
}
|
||||
}
|
||||
```
|
||||
+8
@@ -0,0 +1,8 @@
|
||||
```python
|
||||
def check_gate():
|
||||
# compare this pipeline's constants against what the collection records about itself
|
||||
meta = client.get_collection(COLLECTION).config.metadata or {}
|
||||
|
||||
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
||||
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
||||
```
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
```rust
|
||||
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
let meta = client
|
||||
.collection_info(COLLECTION)
|
||||
.await?
|
||||
.result
|
||||
.and_then(|info| info.config)
|
||||
.map(|config| config.metadata)
|
||||
.unwrap_or_default();
|
||||
|
||||
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|
||||
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
|
||||
!= Some(PIPELINE)
|
||||
{
|
||||
anyhow::bail!(
|
||||
"collection was built by {meta:?}: full re-embed into a fresh collection required"
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
```
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
```typescript
|
||||
async function checkGate() {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
|
||||
{}) as Record<string, unknown>;
|
||||
|
||||
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
|
||||
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
|
||||
}
|
||||
}
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```csharp
|
||||
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
|
||||
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
|
||||
|
||||
var client = new QdrantClient(
|
||||
host: QDRANT_URL!,
|
||||
https: true,
|
||||
apiKey: QDRANT_API_KEY
|
||||
);
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```go
|
||||
QDRANT_URL := os.Getenv("QDRANT_URL")
|
||||
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
|
||||
|
||||
client, err := qdrant.NewClient(&qdrant.Config{
|
||||
Host: QDRANT_URL,
|
||||
APIKey: QDRANT_API_KEY,
|
||||
UseTLS: true,
|
||||
})
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```java
|
||||
static final String QDRANT_URL = System.getenv("QDRANT_URL");
|
||||
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
|
||||
|
||||
static final QdrantClient client =
|
||||
new QdrantClient(
|
||||
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
|
||||
.withApiKey(QDRANT_API_KEY)
|
||||
.build());
|
||||
```
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
```python
|
||||
import os
|
||||
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
QDRANT_URL = os.getenv("QDRANT_URL")
|
||||
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
||||
|
||||
client = QdrantClient(
|
||||
url=QDRANT_URL,
|
||||
api_key=QDRANT_API_KEY,
|
||||
cloud_inference=True
|
||||
)
|
||||
```
|
||||
+8
@@ -0,0 +1,8 @@
|
||||
```rust
|
||||
let qdrant_url = std::env::var("QDRANT_URL")?;
|
||||
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
|
||||
|
||||
let client = Qdrant::from_url(&qdrant_url)
|
||||
.api_key(qdrant_api_key)
|
||||
.build()?;
|
||||
```
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
```typescript
|
||||
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
|
||||
|
||||
const QDRANT_URL = process.env.QDRANT_URL;
|
||||
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
|
||||
|
||||
const client = new QdrantClient({
|
||||
url: QDRANT_URL,
|
||||
apiKey: QDRANT_API_KEY,
|
||||
});
|
||||
```
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
```csharp
|
||||
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
var PIPELINE = "docs-prep-pipeline-v1";
|
||||
var COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
await client.CreateCollectionAsync(
|
||||
collectionName: COLLECTION,
|
||||
vectorsConfig: new VectorParams
|
||||
{
|
||||
Size = 384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance = Distance.Cosine
|
||||
},
|
||||
metadata: new()
|
||||
{
|
||||
["embedding_model"] = MODEL,
|
||||
["pipeline_version"] = PIPELINE
|
||||
}
|
||||
);
|
||||
```
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
```go
|
||||
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
|
||||
PIPELINE := "docs-prep-pipeline-v1"
|
||||
COLLECTION := "docs-sync-tutorial"
|
||||
|
||||
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
CollectionName: COLLECTION,
|
||||
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||
Size: 384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance: qdrant.Distance_Cosine,
|
||||
}),
|
||||
Metadata: qdrant.NewValueMap(map[string]any{
|
||||
"embedding_model": MODEL,
|
||||
"pipeline_version": PIPELINE,
|
||||
}),
|
||||
})
|
||||
```
|
||||
+24
@@ -0,0 +1,24 @@
|
||||
```java
|
||||
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
static final String PIPELINE = "docs-prep-pipeline-v1";
|
||||
static final String COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
static void createCollection() throws Exception {
|
||||
client.createCollectionAsync(
|
||||
CreateCollection.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setVectorsConfig(
|
||||
VectorsConfig.newBuilder()
|
||||
.setParams(
|
||||
VectorParams.newBuilder()
|
||||
.setSize(384) // all-MiniLM-L6-v2 output dimension
|
||||
.setDistance(Distance.Cosine)
|
||||
.build())
|
||||
.build())
|
||||
.putAllMetadata(
|
||||
Map.of(
|
||||
"embedding_model", value(MODEL),
|
||||
"pipeline_version", value(PIPELINE)))
|
||||
.build()).get();
|
||||
}
|
||||
```
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
```python
|
||||
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
||||
PIPELINE = "docs-prep-pipeline-v1"
|
||||
COLLECTION = "docs-sync-tutorial"
|
||||
|
||||
client.create_collection(
|
||||
COLLECTION,
|
||||
vectors_config=models.VectorParams(
|
||||
size=384, # all-MiniLM-L6-v2 output dimension
|
||||
distance=models.Distance.COSINE,
|
||||
),
|
||||
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
||||
)
|
||||
```
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
```rust
|
||||
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
const PIPELINE: &str = "docs-prep-pipeline-v1";
|
||||
const COLLECTION: &str = "docs-sync-tutorial";
|
||||
|
||||
let mut metadata: HashMap<String, Value> = HashMap::new();
|
||||
metadata.insert("embedding_model".to_string(), json!(MODEL));
|
||||
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
|
||||
|
||||
client
|
||||
.create_collection(
|
||||
CreateCollectionBuilder::new(COLLECTION)
|
||||
.vectors_config(VectorParamsBuilder::new(
|
||||
384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance::Cosine,
|
||||
))
|
||||
.metadata(metadata),
|
||||
)
|
||||
.await?;
|
||||
```
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
```typescript
|
||||
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
const PIPELINE = "docs-prep-pipeline-v1";
|
||||
const COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
await client.createCollection(COLLECTION, {
|
||||
vectors: {
|
||||
size: 384, // all-MiniLM-L6-v2 output dimension
|
||||
distance: "Cosine",
|
||||
},
|
||||
});
|
||||
|
||||
await client.updateCollection(COLLECTION, {
|
||||
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
|
||||
});
|
||||
```
|
||||
+248
@@ -0,0 +1,248 @@
|
||||
```csharp
|
||||
using System.Security.Cryptography;
|
||||
using System.Text;
|
||||
using System.Text.RegularExpressions;
|
||||
using Qdrant.Client;
|
||||
using Qdrant.Client.Grpc;
|
||||
using static Qdrant.Client.Grpc.Conditions;
|
||||
using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
|
||||
|
||||
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
|
||||
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
|
||||
|
||||
var client = new QdrantClient(
|
||||
host: QDRANT_URL!,
|
||||
https: true,
|
||||
apiKey: QDRANT_API_KEY
|
||||
);
|
||||
|
||||
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
var PIPELINE = "docs-prep-pipeline-v1";
|
||||
var COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
await client.CreateCollectionAsync(
|
||||
collectionName: COLLECTION,
|
||||
vectorsConfig: new VectorParams
|
||||
{
|
||||
Size = 384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance = Distance.Cosine
|
||||
},
|
||||
metadata: new()
|
||||
{
|
||||
["embedding_model"] = MODEL,
|
||||
["pipeline_version"] = PIPELINE
|
||||
}
|
||||
);
|
||||
|
||||
async Task CheckGate()
|
||||
{
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
|
||||
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
|
||||
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
|
||||
|
||||
if (model != MODEL || pipeline != PIPELINE)
|
||||
throw new InvalidOperationException(
|
||||
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
|
||||
}
|
||||
|
||||
string ContentHash(string text) =>
|
||||
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
|
||||
|
||||
// Qdrant accepts any well-formed UUID as a point ID:
|
||||
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
|
||||
string PointIdFor(string url, string anchor, int num) =>
|
||||
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
|
||||
{
|
||||
var prepared = new List<Chunk>();
|
||||
foreach (var c in chunks)
|
||||
{
|
||||
var text = Normalize(c.Text);
|
||||
prepared.Add(c with
|
||||
{
|
||||
Text = text,
|
||||
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
|
||||
ContentHash = ContentHash(text),
|
||||
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
|
||||
});
|
||||
}
|
||||
return prepared;
|
||||
}
|
||||
|
||||
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
|
||||
{
|
||||
["url"] = chunk.Url,
|
||||
["anchor"] = chunk.Anchor,
|
||||
["chunk_num"] = chunk.ChunkNum,
|
||||
["section_url"] = chunk.SectionUrl,
|
||||
["text"] = chunk.Text,
|
||||
["content_hash"] = chunk.ContentHash,
|
||||
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
|
||||
};
|
||||
|
||||
foreach (var field in new[] { "content_hash", "url", "section_url" })
|
||||
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
|
||||
|
||||
await client.UpsertAsync(
|
||||
collectionName: COLLECTION,
|
||||
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
}).ToList(),
|
||||
wait: true
|
||||
);
|
||||
|
||||
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
await client.QueryAsync(
|
||||
collectionName: COLLECTION,
|
||||
query: new Document { Text = QUERY, Model = MODEL },
|
||||
limit: 3,
|
||||
payloadSelector: new[] { "section_url", "text" }
|
||||
);
|
||||
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
|
||||
SplitByState(List<Chunk> latestChunks)
|
||||
{
|
||||
var incoming = latestChunks.ToDictionary(c => c.PointId);
|
||||
|
||||
var stored = new Dictionary<string, string>();
|
||||
var points = await client.RetrieveAsync(
|
||||
COLLECTION,
|
||||
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
|
||||
payloadSelector: new[] { "content_hash" },
|
||||
vectorSelector: false
|
||||
);
|
||||
foreach (var p in points)
|
||||
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
|
||||
|
||||
var unchanged = new List<Chunk>();
|
||||
var contentChanged = new List<Chunk>();
|
||||
var unknownIds = new List<Chunk>();
|
||||
foreach (var (pid, c) in incoming)
|
||||
{
|
||||
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
|
||||
unchanged.Add(c);
|
||||
else if (stored.ContainsKey(pid))
|
||||
contentChanged.Add(c);
|
||||
else
|
||||
unknownIds.Add(c);
|
||||
}
|
||||
|
||||
return (incoming, unchanged, contentChanged, unknownIds);
|
||||
}
|
||||
|
||||
var splitState = await SplitByState(LATEST_CHUNKS);
|
||||
|
||||
async Task ReEmbedChanged(List<Chunk> contentChanged)
|
||||
{
|
||||
if (contentChanged.Count == 0)
|
||||
return;
|
||||
await client.UpsertAsync(
|
||||
collectionName: COLLECTION,
|
||||
points: contentChanged.Select(c => new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
}).ToList(),
|
||||
wait: true
|
||||
);
|
||||
}
|
||||
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
|
||||
{
|
||||
int reused = 0, added = 0;
|
||||
|
||||
foreach (var c in unknownIds)
|
||||
{
|
||||
var sameText = new Filter
|
||||
{
|
||||
Must = { MatchKeyword("content_hash", c.ContentHash) }
|
||||
};
|
||||
var hits = (await client.ScrollAsync(
|
||||
COLLECTION,
|
||||
filter: sameText,
|
||||
limit: 1,
|
||||
payloadSelector: new[] { "last_updated" },
|
||||
vectorsSelector: true
|
||||
)).Result;
|
||||
|
||||
PointStruct point;
|
||||
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
|
||||
{
|
||||
point = new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
|
||||
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
|
||||
};
|
||||
reused++;
|
||||
}
|
||||
else // genuinely new content: embed and insert
|
||||
{
|
||||
point = new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
};
|
||||
added++;
|
||||
}
|
||||
|
||||
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
|
||||
}
|
||||
|
||||
return (reused, added);
|
||||
}
|
||||
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
|
||||
{
|
||||
if (incomingIds.Count == 0)
|
||||
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
|
||||
|
||||
var stale = new Filter
|
||||
{
|
||||
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
|
||||
};
|
||||
|
||||
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
|
||||
return toDelete;
|
||||
}
|
||||
|
||||
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
|
||||
{
|
||||
await CheckGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
var chunks = PrepareChunksForSync(latestChunks);
|
||||
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
|
||||
|
||||
await ReEmbedChanged(contentChanged);
|
||||
var (reused, added) = await ReuseOrAdd(unknownIds);
|
||||
var deleted = await DeleteGone(incomingIds);
|
||||
|
||||
return new Dictionary<string, long>
|
||||
{
|
||||
["unchanged"] = unchanged.Count,
|
||||
["re-embedded"] = contentChanged.Count,
|
||||
["reused_embedding"] = reused,
|
||||
["added"] = added,
|
||||
["deleted"] = (long)deleted,
|
||||
};
|
||||
}
|
||||
|
||||
var run = await Sync(LATEST_CHUNKS);
|
||||
foreach (var (op, count) in run)
|
||||
Console.WriteLine($"{op}: {count}");
|
||||
```
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
```csharp
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
|
||||
{
|
||||
if (incomingIds.Count == 0)
|
||||
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
|
||||
|
||||
var stale = new Filter
|
||||
{
|
||||
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
|
||||
};
|
||||
|
||||
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
|
||||
return toDelete;
|
||||
}
|
||||
```
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
```go
|
||||
// remove every point the current crawl no longer contains, return how many
|
||||
deleteGone := func(incomingIDs map[string]Chunk) int {
|
||||
if len(incomingIDs) == 0 {
|
||||
panic("Refusing to delete from an empty source snapshot.")
|
||||
}
|
||||
|
||||
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
|
||||
for pid := range incomingIDs {
|
||||
ids = append(ids, qdrant.NewID(pid))
|
||||
}
|
||||
stale := &qdrant.Filter{
|
||||
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
|
||||
}
|
||||
|
||||
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Filter: stale,
|
||||
})
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.Delete(context.Background(), &qdrant.DeletePoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: qdrant.NewPointsSelectorFilter(stale),
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
return int(toDelete)
|
||||
}
|
||||
```
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
```java
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
|
||||
if (incomingIds.isEmpty()) {
|
||||
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
Filter stale = Filter.newBuilder()
|
||||
.addMustNot(hasId(
|
||||
incomingIds.keySet().stream()
|
||||
.map(pid -> id(UUID.fromString(pid)))
|
||||
.collect(Collectors.toList())))
|
||||
.build();
|
||||
|
||||
long toDelete = client.countAsync(COLLECTION, stale, true).get();
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.deleteAsync(COLLECTION, stale).get();
|
||||
return toDelete;
|
||||
}
|
||||
```
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
```python
|
||||
def delete_gone(incoming_ids):
|
||||
"""Remove every point the current crawl no longer contains. Returns how many."""
|
||||
if not incoming_ids:
|
||||
raise ValueError("Refusing to delete from an empty source snapshot.")
|
||||
|
||||
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
||||
|
||||
to_delete = client.count(COLLECTION, count_filter=stale).count
|
||||
|
||||
# potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
||||
return to_delete
|
||||
```
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
```rust
|
||||
/// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async fn delete_gone(
|
||||
client: &Qdrant,
|
||||
incoming_ids: &HashMap<String, Chunk>,
|
||||
) -> anyhow::Result<u64> {
|
||||
if incoming_ids.is_empty() {
|
||||
anyhow::bail!("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
let stale = Filter::must_not([Condition::has_id(
|
||||
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
|
||||
)]);
|
||||
|
||||
let to_delete = client
|
||||
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
|
||||
.await?
|
||||
.result
|
||||
.map(|r| r.count)
|
||||
.unwrap_or(0);
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client
|
||||
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
|
||||
.await?;
|
||||
Ok(to_delete)
|
||||
}
|
||||
```
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
```typescript
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async function deleteGone(incoming: Map<string, SyncChunk>) {
|
||||
if (incoming.size === 0) {
|
||||
throw new Error("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
|
||||
|
||||
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
await client.delete(COLLECTION, { filter: stale, wait: true });
|
||||
return toDelete;
|
||||
}
|
||||
```
|
||||
+276
@@ -0,0 +1,276 @@
|
||||
```go
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"os"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"github.com/qdrant/go-client/qdrant"
|
||||
)
|
||||
|
||||
QDRANT_URL := os.Getenv("QDRANT_URL")
|
||||
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
|
||||
|
||||
client, err := qdrant.NewClient(&qdrant.Config{
|
||||
Host: QDRANT_URL,
|
||||
APIKey: QDRANT_API_KEY,
|
||||
UseTLS: true,
|
||||
})
|
||||
|
||||
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
|
||||
PIPELINE := "docs-prep-pipeline-v1"
|
||||
COLLECTION := "docs-sync-tutorial"
|
||||
|
||||
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
CollectionName: COLLECTION,
|
||||
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||
Size: 384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance: qdrant.Distance_Cosine,
|
||||
}),
|
||||
Metadata: qdrant.NewValueMap(map[string]any{
|
||||
"embedding_model": MODEL,
|
||||
"pipeline_version": PIPELINE,
|
||||
}),
|
||||
})
|
||||
|
||||
checkGate := func() {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
|
||||
meta := info.GetConfig().GetMetadata()
|
||||
|
||||
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
|
||||
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
|
||||
}
|
||||
}
|
||||
|
||||
contentHash := func(text string) string {
|
||||
sum := sha256.Sum256([]byte(text))
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
pointID := func(url, anchor string, num int) string {
|
||||
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
|
||||
// marking the input as a URL-like name
|
||||
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
|
||||
}
|
||||
|
||||
// derive both values (and the section address) for every raw chunk
|
||||
prepareChunksForSync := func(chunks []Chunk) []Chunk {
|
||||
out := make([]Chunk, 0, len(chunks))
|
||||
for _, c := range chunks {
|
||||
c.Text = normalize(c.Text)
|
||||
c.SectionURL = c.URL
|
||||
if c.Anchor != "" {
|
||||
c.SectionURL = c.URL + "#" + c.Anchor
|
||||
}
|
||||
c.ContentHash = contentHash(c.Text)
|
||||
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
|
||||
out = append(out, c)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
payload := func(c Chunk, lastUpdated string) map[string]any {
|
||||
if lastUpdated == "" {
|
||||
lastUpdated = time.Now().UTC().Format(time.RFC3339)
|
||||
}
|
||||
return map[string]any{
|
||||
"url": c.URL,
|
||||
"anchor": c.Anchor,
|
||||
"chunk_num": c.ChunkNum,
|
||||
"section_url": c.SectionURL,
|
||||
"text": c.Text,
|
||||
"content_hash": c.ContentHash,
|
||||
"last_updated": lastUpdated,
|
||||
}
|
||||
}
|
||||
|
||||
for _, field := range []string{"content_hash", "url", "section_url"} {
|
||||
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
|
||||
CollectionName: COLLECTION,
|
||||
FieldName: field,
|
||||
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
|
||||
})
|
||||
}
|
||||
|
||||
var points []*qdrant.PointStruct
|
||||
for _, c := range prepareChunksForSync(CHUNKS) {
|
||||
points = append(points, &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
})
|
||||
}
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: points,
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
|
||||
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||
|
||||
client.Query(context.Background(), &qdrant.QueryPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
|
||||
Limit: qdrant.PtrOf(uint64(3)),
|
||||
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
|
||||
})
|
||||
|
||||
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
|
||||
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
|
||||
incoming := make(map[string]Chunk, len(latestChunks))
|
||||
ids := make([]*qdrant.PointId, 0, len(latestChunks))
|
||||
for _, c := range latestChunks {
|
||||
incoming[c.PointID] = c
|
||||
ids = append(ids, qdrant.NewID(c.PointID))
|
||||
}
|
||||
|
||||
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Ids: ids,
|
||||
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
|
||||
WithVectors: qdrant.NewWithVectors(false),
|
||||
})
|
||||
stored := make(map[string]string, len(retrieved))
|
||||
for _, p := range retrieved {
|
||||
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
|
||||
}
|
||||
|
||||
var unchanged, contentChanged, unknownIDs []Chunk
|
||||
for pid, c := range incoming {
|
||||
storedHash, found := stored[pid]
|
||||
switch {
|
||||
case found && storedHash == c.ContentHash:
|
||||
unchanged = append(unchanged, c)
|
||||
case found:
|
||||
contentChanged = append(contentChanged, c)
|
||||
default:
|
||||
unknownIDs = append(unknownIDs, c)
|
||||
}
|
||||
}
|
||||
|
||||
return incoming, unchanged, contentChanged, unknownIDs
|
||||
}
|
||||
|
||||
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
|
||||
|
||||
reEmbedChanged := func(contentChanged []Chunk) {
|
||||
if len(contentChanged) == 0 {
|
||||
return
|
||||
}
|
||||
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
|
||||
for _, c := range contentChanged {
|
||||
points = append(points, &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
})
|
||||
}
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: points,
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
}
|
||||
|
||||
// reuse an existing embedding when the same text is already stored; embed only what is new
|
||||
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
|
||||
reused, added := 0, 0
|
||||
|
||||
for _, c := range unknownIDs {
|
||||
sameText := &qdrant.Filter{
|
||||
Must: []*qdrant.Condition{
|
||||
qdrant.NewMatch("content_hash", c.ContentHash),
|
||||
},
|
||||
}
|
||||
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Filter: sameText,
|
||||
Limit: qdrant.PtrOf(uint32(1)),
|
||||
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
|
||||
WithVectors: qdrant.NewWithVectors(true),
|
||||
})
|
||||
|
||||
var point *qdrant.PointStruct
|
||||
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
|
||||
point = &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
|
||||
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
|
||||
}
|
||||
reused++
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
}
|
||||
added++
|
||||
}
|
||||
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: []*qdrant.PointStruct{point},
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
}
|
||||
|
||||
return reused, added
|
||||
}
|
||||
|
||||
// remove every point the current crawl no longer contains, return how many
|
||||
deleteGone := func(incomingIDs map[string]Chunk) int {
|
||||
if len(incomingIDs) == 0 {
|
||||
panic("Refusing to delete from an empty source snapshot.")
|
||||
}
|
||||
|
||||
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
|
||||
for pid := range incomingIDs {
|
||||
ids = append(ids, qdrant.NewID(pid))
|
||||
}
|
||||
stale := &qdrant.Filter{
|
||||
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
|
||||
}
|
||||
|
||||
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Filter: stale,
|
||||
})
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.Delete(context.Background(), &qdrant.DeletePoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: qdrant.NewPointsSelectorFilter(stale),
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
return int(toDelete)
|
||||
}
|
||||
|
||||
sync := func(latestChunks []Chunk) map[string]int {
|
||||
checkGate() // refuse to mix embedding models or pipeline versions
|
||||
|
||||
chunks := prepareChunksForSync(latestChunks)
|
||||
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
|
||||
|
||||
reEmbedChanged(contentChanged)
|
||||
reused, added := reuseOrAdd(unknownIDs)
|
||||
deleted := deleteGone(incomingIDs)
|
||||
|
||||
return map[string]int{
|
||||
"unchanged": len(unchanged),
|
||||
"re-embedded": len(contentChanged),
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
}
|
||||
}
|
||||
|
||||
run := sync(LATEST_CHUNKS)
|
||||
fmt.Println(run)
|
||||
```
|
||||
+27
@@ -0,0 +1,27 @@
|
||||
```csharp
|
||||
string ContentHash(string text) =>
|
||||
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
|
||||
|
||||
// Qdrant accepts any well-formed UUID as a point ID:
|
||||
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
|
||||
string PointIdFor(string url, string anchor, int num) =>
|
||||
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
|
||||
{
|
||||
var prepared = new List<Chunk>();
|
||||
foreach (var c in chunks)
|
||||
{
|
||||
var text = Normalize(c.Text);
|
||||
prepared.Add(c with
|
||||
{
|
||||
Text = text,
|
||||
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
|
||||
ContentHash = ContentHash(text),
|
||||
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
|
||||
});
|
||||
}
|
||||
return prepared;
|
||||
}
|
||||
```
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
```go
|
||||
contentHash := func(text string) string {
|
||||
sum := sha256.Sum256([]byte(text))
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
pointID := func(url, anchor string, num int) string {
|
||||
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
|
||||
// marking the input as a URL-like name
|
||||
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
|
||||
}
|
||||
|
||||
// derive both values (and the section address) for every raw chunk
|
||||
prepareChunksForSync := func(chunks []Chunk) []Chunk {
|
||||
out := make([]Chunk, 0, len(chunks))
|
||||
for _, c := range chunks {
|
||||
c.Text = normalize(c.Text)
|
||||
c.SectionURL = c.URL
|
||||
if c.Anchor != "" {
|
||||
c.SectionURL = c.URL + "#" + c.Anchor
|
||||
}
|
||||
c.ContentHash = contentHash(c.Text)
|
||||
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
|
||||
out = append(out, c)
|
||||
}
|
||||
return out
|
||||
}
|
||||
```
|
||||
+27
@@ -0,0 +1,27 @@
|
||||
```java
|
||||
static String contentHash(String text) throws Exception {
|
||||
byte[] digest = MessageDigest.getInstance("SHA-256")
|
||||
.digest(text.getBytes(StandardCharsets.UTF_8));
|
||||
return String.format("%064x", new BigInteger(1, digest));
|
||||
}
|
||||
|
||||
static String pointId(String url, String anchor, int num) {
|
||||
// name-based UUID (version 3); the same address always yields the same ID
|
||||
return UUID.nameUUIDFromBytes(
|
||||
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
|
||||
}
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
|
||||
List<Chunk> out = new ArrayList<>();
|
||||
for (Chunk c : chunks) {
|
||||
String text = normalize(c.text);
|
||||
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
|
||||
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
|
||||
prepared.contentHash = contentHash(text);
|
||||
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
|
||||
out.add(prepared);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
```
|
||||
+26
@@ -0,0 +1,26 @@
|
||||
```python
|
||||
import hashlib
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
|
||||
def content_hash(text):
|
||||
return hashlib.sha256(text.encode()).hexdigest()
|
||||
|
||||
def point_id(url, anchor, num):
|
||||
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
||||
|
||||
def prepare_chunks_for_sync(chunks):
|
||||
"""Derive both values (and the section address) for every raw chunk."""
|
||||
out = []
|
||||
for c in chunks:
|
||||
text = normalize(c["text"])
|
||||
out.append({
|
||||
**c,
|
||||
"text": text,
|
||||
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
||||
"content_hash": content_hash(text),
|
||||
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
||||
})
|
||||
return out
|
||||
```
|
||||
+38
@@ -0,0 +1,38 @@
|
||||
```rust
|
||||
fn content_hash(text: &str) -> String {
|
||||
Sha256::digest(text.as_bytes())
|
||||
.iter()
|
||||
.map(|byte| format!("{byte:02x}"))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn point_id(url: &str, anchor: &str, num: u32) -> String {
|
||||
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||
uuid::Uuid::new_v5(
|
||||
&uuid::Uuid::NAMESPACE_URL,
|
||||
format!("{url}#{anchor}::{num}").as_bytes(),
|
||||
)
|
||||
.to_string()
|
||||
}
|
||||
|
||||
/// Derive both values (and the section address) for every raw chunk.
|
||||
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
|
||||
chunks
|
||||
.iter()
|
||||
.map(|c| {
|
||||
let text = normalize(&c.text);
|
||||
Chunk {
|
||||
text: text.clone(),
|
||||
section_url: if c.anchor.is_empty() {
|
||||
c.url.clone()
|
||||
} else {
|
||||
format!("{}#{}", c.url, c.anchor)
|
||||
},
|
||||
content_hash: content_hash(&text),
|
||||
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
|
||||
..c.clone()
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
```
|
||||
+32
@@ -0,0 +1,32 @@
|
||||
```typescript
|
||||
import { createHash } from "node:crypto";
|
||||
|
||||
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
|
||||
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
|
||||
|
||||
function contentHash(text: string): string {
|
||||
return createHash("sha256").update(text).digest("hex");
|
||||
}
|
||||
|
||||
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
|
||||
function pointId(url: string, anchor: string, num: number): string {
|
||||
// Qdrant accepts any well-formed UUID as a point ID:
|
||||
// hash the address, format the digest as a UUID, and the same address always yields the same ID
|
||||
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
|
||||
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
|
||||
}
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
|
||||
return chunks.map((c) => {
|
||||
const text = normalize(c.text);
|
||||
return {
|
||||
...c,
|
||||
text,
|
||||
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
|
||||
content_hash: contentHash(text),
|
||||
point_id: pointId(c.url, c.anchor, c.chunk_num),
|
||||
};
|
||||
});
|
||||
}
|
||||
```
|
||||
+328
@@ -0,0 +1,328 @@
|
||||
```java
|
||||
import static io.qdrant.client.ConditionFactory.hasId;
|
||||
import static io.qdrant.client.ConditionFactory.matchKeyword;
|
||||
import static io.qdrant.client.PointIdFactory.id;
|
||||
import static io.qdrant.client.QueryFactory.nearest;
|
||||
import static io.qdrant.client.ValueFactory.value;
|
||||
import static io.qdrant.client.VectorFactory.vector;
|
||||
import static io.qdrant.client.VectorsFactory.vectors;
|
||||
|
||||
import io.qdrant.client.QdrantClient;
|
||||
import io.qdrant.client.QdrantGrpcClient;
|
||||
import io.qdrant.client.VectorOutputHelper;
|
||||
import io.qdrant.client.WithPayloadSelectorFactory;
|
||||
import io.qdrant.client.WithVectorsSelectorFactory;
|
||||
import io.qdrant.client.grpc.Collections.CreateCollection;
|
||||
import io.qdrant.client.grpc.Collections.Distance;
|
||||
import io.qdrant.client.grpc.Collections.PayloadSchemaType;
|
||||
import io.qdrant.client.grpc.Collections.VectorParams;
|
||||
import io.qdrant.client.grpc.Collections.VectorsConfig;
|
||||
import io.qdrant.client.grpc.Common.Filter;
|
||||
import io.qdrant.client.grpc.JsonWithInt.Value;
|
||||
import io.qdrant.client.grpc.Points.Document;
|
||||
import io.qdrant.client.grpc.Points.PointStruct;
|
||||
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||
import io.qdrant.client.grpc.Points.ScrollPoints;
|
||||
import java.math.BigInteger;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.time.OffsetDateTime;
|
||||
import java.time.ZoneOffset;
|
||||
import java.time.temporal.ChronoUnit;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.UUID;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
static final String QDRANT_URL = System.getenv("QDRANT_URL");
|
||||
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
|
||||
|
||||
static final QdrantClient client =
|
||||
new QdrantClient(
|
||||
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
|
||||
.withApiKey(QDRANT_API_KEY)
|
||||
.build());
|
||||
|
||||
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
static final String PIPELINE = "docs-prep-pipeline-v1";
|
||||
static final String COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
static void createCollection() throws Exception {
|
||||
client.createCollectionAsync(
|
||||
CreateCollection.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setVectorsConfig(
|
||||
VectorsConfig.newBuilder()
|
||||
.setParams(
|
||||
VectorParams.newBuilder()
|
||||
.setSize(384) // all-MiniLM-L6-v2 output dimension
|
||||
.setDistance(Distance.Cosine)
|
||||
.build())
|
||||
.build())
|
||||
.putAllMetadata(
|
||||
Map.of(
|
||||
"embedding_model", value(MODEL),
|
||||
"pipeline_version", value(PIPELINE)))
|
||||
.build()).get();
|
||||
}
|
||||
|
||||
static void checkGate() throws Exception {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
Map<String, Value> meta =
|
||||
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
|
||||
|
||||
Value model = meta.get("embedding_model");
|
||||
Value pipeline = meta.get("pipeline_version");
|
||||
if (model == null || !MODEL.equals(model.getStringValue())
|
||||
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
|
||||
throw new RuntimeException(
|
||||
"collection was built by " + meta + ": full re-embed into a fresh collection required");
|
||||
}
|
||||
}
|
||||
|
||||
static String contentHash(String text) throws Exception {
|
||||
byte[] digest = MessageDigest.getInstance("SHA-256")
|
||||
.digest(text.getBytes(StandardCharsets.UTF_8));
|
||||
return String.format("%064x", new BigInteger(1, digest));
|
||||
}
|
||||
|
||||
static String pointId(String url, String anchor, int num) {
|
||||
// name-based UUID (version 3); the same address always yields the same ID
|
||||
return UUID.nameUUIDFromBytes(
|
||||
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
|
||||
}
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
|
||||
List<Chunk> out = new ArrayList<>();
|
||||
for (Chunk c : chunks) {
|
||||
String text = normalize(c.text);
|
||||
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
|
||||
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
|
||||
prepared.contentHash = contentHash(text);
|
||||
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
|
||||
out.add(prepared);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
|
||||
Map<String, Value> p = new HashMap<>();
|
||||
p.put("url", value(chunk.url));
|
||||
p.put("anchor", value(chunk.anchor));
|
||||
p.put("chunk_num", value(chunk.chunkNum));
|
||||
p.put("section_url", value(chunk.sectionUrl));
|
||||
p.put("text", value(chunk.text));
|
||||
p.put("content_hash", value(chunk.contentHash));
|
||||
p.put("last_updated", value(lastUpdated != null
|
||||
? lastUpdated
|
||||
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
|
||||
return p;
|
||||
}
|
||||
|
||||
static void createPayloadIndexes() throws Exception {
|
||||
for (String field : List.of("content_hash", "url", "section_url")) {
|
||||
client.createPayloadIndexAsync(
|
||||
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
|
||||
}
|
||||
}
|
||||
|
||||
static void populate() throws Exception {
|
||||
List<PointStruct> points = new ArrayList<>();
|
||||
for (Chunk c : prepareChunksForSync(CHUNKS)) {
|
||||
points.add(
|
||||
PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build());
|
||||
}
|
||||
client.upsertAsync(COLLECTION, points).get();
|
||||
}
|
||||
|
||||
static final String QUERY =
|
||||
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
static void search() throws Exception {
|
||||
client.queryAsync(
|
||||
QueryPoints.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setQuery(
|
||||
nearest(
|
||||
Document.newBuilder()
|
||||
.setText(QUERY)
|
||||
.setModel(MODEL)
|
||||
.build()))
|
||||
.setLimit(3)
|
||||
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
|
||||
.build()).get();
|
||||
}
|
||||
|
||||
static class SyncState {
|
||||
Map<String, Chunk> incoming = new LinkedHashMap<>();
|
||||
List<Chunk> unchanged = new ArrayList<>();
|
||||
List<Chunk> contentChanged = new ArrayList<>();
|
||||
List<Chunk> unknownIds = new ArrayList<>();
|
||||
}
|
||||
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
|
||||
SyncState state = new SyncState();
|
||||
for (Chunk c : latestChunks) {
|
||||
state.incoming.put(c.pointId, c);
|
||||
}
|
||||
|
||||
Map<String, String> stored = new HashMap<>();
|
||||
var points = client.retrieveAsync(
|
||||
COLLECTION,
|
||||
state.incoming.keySet().stream()
|
||||
.map(pid -> id(UUID.fromString(pid)))
|
||||
.collect(Collectors.toList()),
|
||||
WithPayloadSelectorFactory.include(List.of("content_hash")),
|
||||
WithVectorsSelectorFactory.enable(false),
|
||||
null).get();
|
||||
for (var p : points) {
|
||||
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
|
||||
}
|
||||
|
||||
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
|
||||
String pid = e.getKey();
|
||||
Chunk c = e.getValue();
|
||||
if (c.contentHash.equals(stored.get(pid))) {
|
||||
state.unchanged.add(c);
|
||||
} else if (stored.containsKey(pid)) {
|
||||
state.contentChanged.add(c);
|
||||
} else {
|
||||
state.unknownIds.add(c);
|
||||
}
|
||||
}
|
||||
|
||||
return state;
|
||||
}
|
||||
|
||||
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
|
||||
if (contentChanged.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
List<PointStruct> points = new ArrayList<>();
|
||||
for (Chunk c : contentChanged) {
|
||||
points.add(
|
||||
PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build());
|
||||
}
|
||||
client.upsertAsync(COLLECTION, points).get();
|
||||
}
|
||||
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
|
||||
int reused = 0;
|
||||
int added = 0;
|
||||
|
||||
for (Chunk c : unknownIds) {
|
||||
Filter sameText = Filter.newBuilder()
|
||||
.addMust(matchKeyword("content_hash", c.contentHash))
|
||||
.build();
|
||||
|
||||
var hits = client.scrollAsync(
|
||||
ScrollPoints.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setFilter(sameText)
|
||||
.setLimit(1)
|
||||
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
|
||||
.setWithVectors(WithVectorsSelectorFactory.enable(true))
|
||||
.build()).get().getResultList();
|
||||
|
||||
PointStruct point;
|
||||
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
|
||||
point = PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(vectors(vector(
|
||||
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
|
||||
.getDataList())))
|
||||
.putAllPayload(
|
||||
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
|
||||
.build();
|
||||
reused++;
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build();
|
||||
added++;
|
||||
}
|
||||
|
||||
client.upsertAsync(COLLECTION, List.of(point)).get();
|
||||
}
|
||||
|
||||
return new int[] {reused, added};
|
||||
}
|
||||
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
|
||||
if (incomingIds.isEmpty()) {
|
||||
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
Filter stale = Filter.newBuilder()
|
||||
.addMustNot(hasId(
|
||||
incomingIds.keySet().stream()
|
||||
.map(pid -> id(UUID.fromString(pid)))
|
||||
.collect(Collectors.toList())))
|
||||
.build();
|
||||
|
||||
long toDelete = client.countAsync(COLLECTION, stale, true).get();
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.deleteAsync(COLLECTION, stale).get();
|
||||
return toDelete;
|
||||
}
|
||||
|
||||
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
|
||||
checkGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
List<Chunk> chunks = prepareChunksForSync(latestChunks);
|
||||
SyncState state = splitByState(chunks);
|
||||
|
||||
reEmbedChanged(state.contentChanged);
|
||||
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
|
||||
long deleted = deleteGone(state.incoming);
|
||||
|
||||
return Map.of(
|
||||
"unchanged", (long) state.unchanged.size(),
|
||||
"re-embedded", (long) state.contentChanged.size(),
|
||||
"reused_embedding", (long) reusedAdded[0],
|
||||
"added", (long) reusedAdded[1],
|
||||
"deleted", deleted);
|
||||
}
|
||||
|
||||
static void runSync() throws Exception {
|
||||
Map<String, Long> run = sync(LATEST_CHUNKS);
|
||||
System.out.println(run);
|
||||
}
|
||||
```
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
```csharp
|
||||
foreach (var field in new[] { "content_hash", "url", "section_url" })
|
||||
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
|
||||
```
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
```go
|
||||
for _, field := range []string{"content_hash", "url", "section_url"} {
|
||||
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
|
||||
CollectionName: COLLECTION,
|
||||
FieldName: field,
|
||||
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
|
||||
})
|
||||
}
|
||||
```
|
||||
+8
@@ -0,0 +1,8 @@
|
||||
```java
|
||||
static void createPayloadIndexes() throws Exception {
|
||||
for (String field : List.of("content_hash", "url", "section_url")) {
|
||||
client.createPayloadIndexAsync(
|
||||
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
|
||||
}
|
||||
}
|
||||
```
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
```python
|
||||
for field in ("content_hash", "url", "section_url"):
|
||||
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
||||
```
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
```rust
|
||||
for field in ["content_hash", "url", "section_url"] {
|
||||
client
|
||||
.create_field_index(CreateFieldIndexCollectionBuilder::new(
|
||||
COLLECTION,
|
||||
field,
|
||||
FieldType::Keyword,
|
||||
))
|
||||
.await?;
|
||||
}
|
||||
```
|
||||
+8
@@ -0,0 +1,8 @@
|
||||
```typescript
|
||||
for (const field of ["content_hash", "url", "section_url"]) {
|
||||
await client.createPayloadIndex(COLLECTION, {
|
||||
field_name: field,
|
||||
field_schema: "keyword",
|
||||
});
|
||||
}
|
||||
```
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
```csharp
|
||||
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
|
||||
{
|
||||
["url"] = chunk.Url,
|
||||
["anchor"] = chunk.Anchor,
|
||||
["chunk_num"] = chunk.ChunkNum,
|
||||
["section_url"] = chunk.SectionUrl,
|
||||
["text"] = chunk.Text,
|
||||
["content_hash"] = chunk.ContentHash,
|
||||
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
|
||||
};
|
||||
```
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
```go
|
||||
payload := func(c Chunk, lastUpdated string) map[string]any {
|
||||
if lastUpdated == "" {
|
||||
lastUpdated = time.Now().UTC().Format(time.RFC3339)
|
||||
}
|
||||
return map[string]any{
|
||||
"url": c.URL,
|
||||
"anchor": c.Anchor,
|
||||
"chunk_num": c.ChunkNum,
|
||||
"section_url": c.SectionURL,
|
||||
"text": c.Text,
|
||||
"content_hash": c.ContentHash,
|
||||
"last_updated": lastUpdated,
|
||||
}
|
||||
}
|
||||
```
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
```java
|
||||
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
|
||||
Map<String, Value> p = new HashMap<>();
|
||||
p.put("url", value(chunk.url));
|
||||
p.put("anchor", value(chunk.anchor));
|
||||
p.put("chunk_num", value(chunk.chunkNum));
|
||||
p.put("section_url", value(chunk.sectionUrl));
|
||||
p.put("text", value(chunk.text));
|
||||
p.put("content_hash", value(chunk.contentHash));
|
||||
p.put("last_updated", value(lastUpdated != null
|
||||
? lastUpdated
|
||||
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
|
||||
return p;
|
||||
}
|
||||
```
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
```python
|
||||
def payload(chunk, last_updated=None):
|
||||
return {
|
||||
"url": chunk["url"],
|
||||
"anchor": chunk["anchor"],
|
||||
"chunk_num": chunk["chunk_num"],
|
||||
"section_url": chunk["section_url"],
|
||||
"text": chunk["text"],
|
||||
"content_hash": chunk["content_hash"],
|
||||
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||
}
|
||||
```
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
```rust
|
||||
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
|
||||
let last_updated = last_updated.unwrap_or_else(|| {
|
||||
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
|
||||
});
|
||||
Ok(Payload::try_from(serde_json::json!({
|
||||
"url": chunk.url,
|
||||
"anchor": chunk.anchor,
|
||||
"chunk_num": chunk.chunk_num,
|
||||
"section_url": chunk.section_url,
|
||||
"text": chunk.text,
|
||||
"content_hash": chunk.content_hash,
|
||||
"last_updated": last_updated,
|
||||
}))?)
|
||||
}
|
||||
```
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
```typescript
|
||||
function payload(chunk: SyncChunk, lastUpdated?: string) {
|
||||
return {
|
||||
url: chunk.url,
|
||||
anchor: chunk.anchor,
|
||||
chunk_num: chunk.chunk_num,
|
||||
section_url: chunk.section_url,
|
||||
text: chunk.text,
|
||||
content_hash: chunk.content_hash,
|
||||
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
|
||||
};
|
||||
}
|
||||
```
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
```csharp
|
||||
await client.UpsertAsync(
|
||||
collectionName: COLLECTION,
|
||||
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
}).ToList(),
|
||||
wait: true
|
||||
);
|
||||
```
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
```go
|
||||
var points []*qdrant.PointStruct
|
||||
for _, c := range prepareChunksForSync(CHUNKS) {
|
||||
points = append(points, &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
})
|
||||
}
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: points,
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
```
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
```java
|
||||
static void populate() throws Exception {
|
||||
List<PointStruct> points = new ArrayList<>();
|
||||
for (Chunk c : prepareChunksForSync(CHUNKS)) {
|
||||
points.add(
|
||||
PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build());
|
||||
}
|
||||
client.upsertAsync(COLLECTION, points).get();
|
||||
}
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```python
|
||||
client.upsert(COLLECTION, points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in prepare_chunks_for_sync(CHUNKS)
|
||||
], wait=True)
|
||||
```
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
```rust
|
||||
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
|
||||
.iter()
|
||||
.map(|c| {
|
||||
Ok(PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
))
|
||||
})
|
||||
.collect::<anyhow::Result<_>>()?;
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||
.await?;
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```typescript
|
||||
await client.upsert(COLLECTION, {
|
||||
points: prepareChunksForSync(CHUNKS).map((c) => ({
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
})),
|
||||
wait: true,
|
||||
});
|
||||
```
|
||||
+203
@@ -0,0 +1,203 @@
|
||||
```python
|
||||
import os
|
||||
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
QDRANT_URL = os.getenv("QDRANT_URL")
|
||||
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
||||
|
||||
client = QdrantClient(
|
||||
url=QDRANT_URL,
|
||||
api_key=QDRANT_API_KEY,
|
||||
cloud_inference=True
|
||||
)
|
||||
|
||||
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
||||
PIPELINE = "docs-prep-pipeline-v1"
|
||||
COLLECTION = "docs-sync-tutorial"
|
||||
|
||||
client.create_collection(
|
||||
COLLECTION,
|
||||
vectors_config=models.VectorParams(
|
||||
size=384, # all-MiniLM-L6-v2 output dimension
|
||||
distance=models.Distance.COSINE,
|
||||
),
|
||||
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
||||
)
|
||||
|
||||
def check_gate():
|
||||
# compare this pipeline's constants against what the collection records about itself
|
||||
meta = client.get_collection(COLLECTION).config.metadata or {}
|
||||
|
||||
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
||||
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
||||
|
||||
import hashlib
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
|
||||
def content_hash(text):
|
||||
return hashlib.sha256(text.encode()).hexdigest()
|
||||
|
||||
def point_id(url, anchor, num):
|
||||
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
||||
|
||||
def prepare_chunks_for_sync(chunks):
|
||||
"""Derive both values (and the section address) for every raw chunk."""
|
||||
out = []
|
||||
for c in chunks:
|
||||
text = normalize(c["text"])
|
||||
out.append({
|
||||
**c,
|
||||
"text": text,
|
||||
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
||||
"content_hash": content_hash(text),
|
||||
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
||||
})
|
||||
return out
|
||||
|
||||
def payload(chunk, last_updated=None):
|
||||
return {
|
||||
"url": chunk["url"],
|
||||
"anchor": chunk["anchor"],
|
||||
"chunk_num": chunk["chunk_num"],
|
||||
"section_url": chunk["section_url"],
|
||||
"text": chunk["text"],
|
||||
"content_hash": chunk["content_hash"],
|
||||
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||
}
|
||||
|
||||
for field in ("content_hash", "url", "section_url"):
|
||||
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
||||
|
||||
client.upsert(COLLECTION, points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in prepare_chunks_for_sync(CHUNKS)
|
||||
], wait=True)
|
||||
|
||||
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||
|
||||
client.query_points(
|
||||
COLLECTION,
|
||||
query=models.Document(text=QUERY, model=MODEL),
|
||||
limit=3,
|
||||
with_payload=["section_url", "text"],
|
||||
)
|
||||
|
||||
def split_by_state(latest_chunks):
|
||||
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
||||
incoming = {c["point_id"]: c for c in latest_chunks}
|
||||
|
||||
stored = {}
|
||||
points = client.retrieve(
|
||||
COLLECTION,
|
||||
ids=list(incoming),
|
||||
with_payload=["content_hash"],
|
||||
with_vectors=False,
|
||||
)
|
||||
for p in points:
|
||||
stored[str(p.id)] = p.payload["content_hash"]
|
||||
|
||||
unchanged, content_changed, unknown_ids = [], [], []
|
||||
for pid, c in incoming.items():
|
||||
if stored.get(pid) == c["content_hash"]:
|
||||
unchanged.append(c)
|
||||
elif pid in stored:
|
||||
content_changed.append(c)
|
||||
else:
|
||||
unknown_ids.append(c)
|
||||
|
||||
return incoming, unchanged, content_changed, unknown_ids
|
||||
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
||||
|
||||
def re_embed_changed(content_changed):
|
||||
if not content_changed:
|
||||
return
|
||||
client.upsert(COLLECTION,
|
||||
points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in content_changed],
|
||||
wait=True)
|
||||
|
||||
def reuse_or_add(unknown_ids):
|
||||
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
||||
reused, added = 0, 0
|
||||
|
||||
for c in unknown_ids:
|
||||
same_text = models.Filter(must=[
|
||||
models.FieldCondition(
|
||||
key="content_hash",
|
||||
match=models.MatchValue(value=c["content_hash"]),
|
||||
)
|
||||
])
|
||||
hits, _ = client.scroll(
|
||||
COLLECTION,
|
||||
scroll_filter=same_text,
|
||||
limit=1,
|
||||
with_payload=["last_updated"],
|
||||
with_vectors=True,
|
||||
)
|
||||
|
||||
if hits: # same text, new address: copy the vector, keep its last_updated
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=hits[0].vector,
|
||||
payload=payload(c, hits[0].payload["last_updated"]),
|
||||
)
|
||||
reused += 1
|
||||
else: # genuinely new content: embed and insert
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
added += 1
|
||||
|
||||
client.upsert(COLLECTION, points=[point], wait=True)
|
||||
|
||||
return reused, added
|
||||
|
||||
def delete_gone(incoming_ids):
|
||||
"""Remove every point the current crawl no longer contains. Returns how many."""
|
||||
if not incoming_ids:
|
||||
raise ValueError("Refusing to delete from an empty source snapshot.")
|
||||
|
||||
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
||||
|
||||
to_delete = client.count(COLLECTION, count_filter=stale).count
|
||||
|
||||
# potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
||||
return to_delete
|
||||
|
||||
def sync(latest_chunks):
|
||||
check_gate() # refuse to mix embedding models or pipeline versions
|
||||
|
||||
chunks = prepare_chunks_for_sync(latest_chunks)
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
||||
|
||||
re_embed_changed(content_changed)
|
||||
reused, added = reuse_or_add(unknown_ids)
|
||||
deleted = delete_gone(incoming_ids)
|
||||
|
||||
return {
|
||||
"unchanged": len(unchanged),
|
||||
"re-embedded": len(content_changed),
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
}
|
||||
|
||||
run = sync(LATEST_CHUNKS)
|
||||
print(run)
|
||||
```
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
```csharp
|
||||
async Task ReEmbedChanged(List<Chunk> contentChanged)
|
||||
{
|
||||
if (contentChanged.Count == 0)
|
||||
return;
|
||||
await client.UpsertAsync(
|
||||
collectionName: COLLECTION,
|
||||
points: contentChanged.Select(c => new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
}).ToList(),
|
||||
wait: true
|
||||
);
|
||||
}
|
||||
```
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
```go
|
||||
reEmbedChanged := func(contentChanged []Chunk) {
|
||||
if len(contentChanged) == 0 {
|
||||
return
|
||||
}
|
||||
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
|
||||
for _, c := range contentChanged {
|
||||
points = append(points, &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
})
|
||||
}
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: points,
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
}
|
||||
```
|
||||
+23
@@ -0,0 +1,23 @@
|
||||
```java
|
||||
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
|
||||
if (contentChanged.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
List<PointStruct> points = new ArrayList<>();
|
||||
for (Chunk c : contentChanged) {
|
||||
points.add(
|
||||
PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build());
|
||||
}
|
||||
client.upsertAsync(COLLECTION, points).get();
|
||||
}
|
||||
```
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
```python
|
||||
def re_embed_changed(content_changed):
|
||||
if not content_changed:
|
||||
return
|
||||
client.upsert(COLLECTION,
|
||||
points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in content_changed],
|
||||
wait=True)
|
||||
```
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
```rust
|
||||
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
|
||||
if content_changed.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let points: Vec<PointStruct> = content_changed
|
||||
.iter()
|
||||
.map(|c| {
|
||||
Ok(PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
))
|
||||
})
|
||||
.collect::<anyhow::Result<_>>()?;
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
```
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
```typescript
|
||||
async function reEmbedChanged(contentChanged: SyncChunk[]) {
|
||||
if (contentChanged.length === 0) {
|
||||
return;
|
||||
}
|
||||
await client.upsert(COLLECTION, {
|
||||
points: contentChanged.map((c) => ({
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
})),
|
||||
wait: true,
|
||||
});
|
||||
}
|
||||
```
|
||||
+48
@@ -0,0 +1,48 @@
|
||||
```csharp
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
|
||||
{
|
||||
int reused = 0, added = 0;
|
||||
|
||||
foreach (var c in unknownIds)
|
||||
{
|
||||
var sameText = new Filter
|
||||
{
|
||||
Must = { MatchKeyword("content_hash", c.ContentHash) }
|
||||
};
|
||||
var hits = (await client.ScrollAsync(
|
||||
COLLECTION,
|
||||
filter: sameText,
|
||||
limit: 1,
|
||||
payloadSelector: new[] { "last_updated" },
|
||||
vectorsSelector: true
|
||||
)).Result;
|
||||
|
||||
PointStruct point;
|
||||
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
|
||||
{
|
||||
point = new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
|
||||
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
|
||||
};
|
||||
reused++;
|
||||
}
|
||||
else // genuinely new content: embed and insert
|
||||
{
|
||||
point = new PointStruct
|
||||
{
|
||||
Id = new PointId { Uuid = c.PointId },
|
||||
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||
Payload = { Payload(c) },
|
||||
};
|
||||
added++;
|
||||
}
|
||||
|
||||
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
|
||||
}
|
||||
|
||||
return (reused, added);
|
||||
}
|
||||
```
|
||||
+46
@@ -0,0 +1,46 @@
|
||||
```go
|
||||
// reuse an existing embedding when the same text is already stored; embed only what is new
|
||||
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
|
||||
reused, added := 0, 0
|
||||
|
||||
for _, c := range unknownIDs {
|
||||
sameText := &qdrant.Filter{
|
||||
Must: []*qdrant.Condition{
|
||||
qdrant.NewMatch("content_hash", c.ContentHash),
|
||||
},
|
||||
}
|
||||
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Filter: sameText,
|
||||
Limit: qdrant.PtrOf(uint32(1)),
|
||||
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
|
||||
WithVectors: qdrant.NewWithVectors(true),
|
||||
})
|
||||
|
||||
var point *qdrant.PointStruct
|
||||
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
|
||||
point = &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
|
||||
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
|
||||
}
|
||||
reused++
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
}
|
||||
added++
|
||||
}
|
||||
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: []*qdrant.PointStruct{point},
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
}
|
||||
|
||||
return reused, added
|
||||
}
|
||||
```
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
```java
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
|
||||
int reused = 0;
|
||||
int added = 0;
|
||||
|
||||
for (Chunk c : unknownIds) {
|
||||
Filter sameText = Filter.newBuilder()
|
||||
.addMust(matchKeyword("content_hash", c.contentHash))
|
||||
.build();
|
||||
|
||||
var hits = client.scrollAsync(
|
||||
ScrollPoints.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setFilter(sameText)
|
||||
.setLimit(1)
|
||||
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
|
||||
.setWithVectors(WithVectorsSelectorFactory.enable(true))
|
||||
.build()).get().getResultList();
|
||||
|
||||
PointStruct point;
|
||||
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
|
||||
point = PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(vectors(vector(
|
||||
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
|
||||
.getDataList())))
|
||||
.putAllPayload(
|
||||
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
|
||||
.build();
|
||||
reused++;
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build();
|
||||
added++;
|
||||
}
|
||||
|
||||
client.upsertAsync(COLLECTION, List.of(point)).get();
|
||||
}
|
||||
|
||||
return new int[] {reused, added};
|
||||
}
|
||||
```
|
||||
+39
@@ -0,0 +1,39 @@
|
||||
```python
|
||||
def reuse_or_add(unknown_ids):
|
||||
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
||||
reused, added = 0, 0
|
||||
|
||||
for c in unknown_ids:
|
||||
same_text = models.Filter(must=[
|
||||
models.FieldCondition(
|
||||
key="content_hash",
|
||||
match=models.MatchValue(value=c["content_hash"]),
|
||||
)
|
||||
])
|
||||
hits, _ = client.scroll(
|
||||
COLLECTION,
|
||||
scroll_filter=same_text,
|
||||
limit=1,
|
||||
with_payload=["last_updated"],
|
||||
with_vectors=True,
|
||||
)
|
||||
|
||||
if hits: # same text, new address: copy the vector, keep its last_updated
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=hits[0].vector,
|
||||
payload=payload(c, hits[0].payload["last_updated"]),
|
||||
)
|
||||
reused += 1
|
||||
else: # genuinely new content: embed and insert
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
added += 1
|
||||
|
||||
client.upsert(COLLECTION, points=[point], wait=True)
|
||||
|
||||
return reused, added
|
||||
```
|
||||
+51
@@ -0,0 +1,51 @@
|
||||
```rust
|
||||
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
|
||||
let (mut reused, mut added) = (0, 0);
|
||||
|
||||
for c in unknown_ids {
|
||||
let same_text =
|
||||
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
|
||||
let hits = client
|
||||
.scroll(
|
||||
ScrollPointsBuilder::new(COLLECTION)
|
||||
.filter(same_text)
|
||||
.limit(1)
|
||||
.with_payload(PayloadIncludeSelector::new(vec![
|
||||
"last_updated".to_string()
|
||||
]))
|
||||
.with_vectors(true),
|
||||
)
|
||||
.await?
|
||||
.result;
|
||||
|
||||
let point = if let Some(hit) = hits.into_iter().next() {
|
||||
// same text, new address: copy the vector, keep its last_updated
|
||||
let last_updated = hit.get("last_updated").as_str().cloned();
|
||||
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
|
||||
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
|
||||
Some(vector_output::Vector::Dense(dense)) => dense.data,
|
||||
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||
},
|
||||
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||
};
|
||||
reused += 1;
|
||||
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
|
||||
} else {
|
||||
// genuinely new content: embed and insert
|
||||
added += 1;
|
||||
PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
)
|
||||
};
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
|
||||
.await?;
|
||||
}
|
||||
|
||||
Ok((reused, added))
|
||||
}
|
||||
```
|
||||
+45
@@ -0,0 +1,45 @@
|
||||
```typescript
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async function reuseOrAdd(unknownIds: SyncChunk[]) {
|
||||
let reused = 0;
|
||||
let added = 0;
|
||||
|
||||
for (const c of unknownIds) {
|
||||
const sameText = {
|
||||
must: [
|
||||
{
|
||||
key: "content_hash",
|
||||
match: { value: c.content_hash },
|
||||
},
|
||||
],
|
||||
};
|
||||
const hits = (await client.scroll(COLLECTION, {
|
||||
filter: sameText,
|
||||
limit: 1,
|
||||
with_payload: ["last_updated"],
|
||||
with_vector: true,
|
||||
})).points;
|
||||
|
||||
let point: Schemas["PointStruct"];
|
||||
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
|
||||
point = {
|
||||
id: c.point_id,
|
||||
vector: hits[0].vector as number[],
|
||||
payload: payload(c, hits[0].payload?.last_updated as string),
|
||||
};
|
||||
reused += 1;
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = {
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
};
|
||||
added += 1;
|
||||
}
|
||||
|
||||
await client.upsert(COLLECTION, { points: [point], wait: true });
|
||||
}
|
||||
|
||||
return { reused, added };
|
||||
}
|
||||
```
|
||||
+5
@@ -0,0 +1,5 @@
|
||||
```csharp
|
||||
var run = await Sync(LATEST_CHUNKS);
|
||||
foreach (var (op, count) in run)
|
||||
Console.WriteLine($"{op}: {count}");
|
||||
```
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
```go
|
||||
run := sync(LATEST_CHUNKS)
|
||||
fmt.Println(run)
|
||||
```
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
```java
|
||||
static void runSync() throws Exception {
|
||||
Map<String, Long> run = sync(LATEST_CHUNKS);
|
||||
System.out.println(run);
|
||||
}
|
||||
```
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
```python
|
||||
run = sync(LATEST_CHUNKS)
|
||||
print(run)
|
||||
```
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
```rust
|
||||
let run = sync(&client, &latest_chunks).await?;
|
||||
println!("{run:?}");
|
||||
```
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
```typescript
|
||||
const run = await sync(LATEST_CHUNKS);
|
||||
console.log(run);
|
||||
```
|
||||
+322
@@ -0,0 +1,322 @@
|
||||
```rust
|
||||
use serde_json::{json, Value};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use qdrant_client::qdrant::{
|
||||
point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
|
||||
CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
|
||||
Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
|
||||
Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
|
||||
};
|
||||
use qdrant_client::{Payload, Qdrant};
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
let qdrant_url = std::env::var("QDRANT_URL")?;
|
||||
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
|
||||
|
||||
let client = Qdrant::from_url(&qdrant_url)
|
||||
.api_key(qdrant_api_key)
|
||||
.build()?;
|
||||
|
||||
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
const PIPELINE: &str = "docs-prep-pipeline-v1";
|
||||
const COLLECTION: &str = "docs-sync-tutorial";
|
||||
|
||||
let mut metadata: HashMap<String, Value> = HashMap::new();
|
||||
metadata.insert("embedding_model".to_string(), json!(MODEL));
|
||||
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
|
||||
|
||||
client
|
||||
.create_collection(
|
||||
CreateCollectionBuilder::new(COLLECTION)
|
||||
.vectors_config(VectorParamsBuilder::new(
|
||||
384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance::Cosine,
|
||||
))
|
||||
.metadata(metadata),
|
||||
)
|
||||
.await?;
|
||||
|
||||
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
let meta = client
|
||||
.collection_info(COLLECTION)
|
||||
.await?
|
||||
.result
|
||||
.and_then(|info| info.config)
|
||||
.map(|config| config.metadata)
|
||||
.unwrap_or_default();
|
||||
|
||||
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|
||||
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
|
||||
!= Some(PIPELINE)
|
||||
{
|
||||
anyhow::bail!(
|
||||
"collection was built by {meta:?}: full re-embed into a fresh collection required"
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn content_hash(text: &str) -> String {
|
||||
Sha256::digest(text.as_bytes())
|
||||
.iter()
|
||||
.map(|byte| format!("{byte:02x}"))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn point_id(url: &str, anchor: &str, num: u32) -> String {
|
||||
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||
uuid::Uuid::new_v5(
|
||||
&uuid::Uuid::NAMESPACE_URL,
|
||||
format!("{url}#{anchor}::{num}").as_bytes(),
|
||||
)
|
||||
.to_string()
|
||||
}
|
||||
|
||||
/// Derive both values (and the section address) for every raw chunk.
|
||||
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
|
||||
chunks
|
||||
.iter()
|
||||
.map(|c| {
|
||||
let text = normalize(&c.text);
|
||||
Chunk {
|
||||
text: text.clone(),
|
||||
section_url: if c.anchor.is_empty() {
|
||||
c.url.clone()
|
||||
} else {
|
||||
format!("{}#{}", c.url, c.anchor)
|
||||
},
|
||||
content_hash: content_hash(&text),
|
||||
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
|
||||
..c.clone()
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
|
||||
let last_updated = last_updated.unwrap_or_else(|| {
|
||||
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
|
||||
});
|
||||
Ok(Payload::try_from(serde_json::json!({
|
||||
"url": chunk.url,
|
||||
"anchor": chunk.anchor,
|
||||
"chunk_num": chunk.chunk_num,
|
||||
"section_url": chunk.section_url,
|
||||
"text": chunk.text,
|
||||
"content_hash": chunk.content_hash,
|
||||
"last_updated": last_updated,
|
||||
}))?)
|
||||
}
|
||||
|
||||
for field in ["content_hash", "url", "section_url"] {
|
||||
client
|
||||
.create_field_index(CreateFieldIndexCollectionBuilder::new(
|
||||
COLLECTION,
|
||||
field,
|
||||
FieldType::Keyword,
|
||||
))
|
||||
.await?;
|
||||
}
|
||||
|
||||
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
|
||||
.iter()
|
||||
.map(|c| {
|
||||
Ok(PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
))
|
||||
})
|
||||
.collect::<anyhow::Result<_>>()?;
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||
.await?;
|
||||
|
||||
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
client
|
||||
.query(
|
||||
QueryPointsBuilder::new(COLLECTION)
|
||||
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
|
||||
.limit(3)
|
||||
.with_payload(PayloadIncludeSelector::new(vec![
|
||||
"section_url".to_string(),
|
||||
"text".to_string(),
|
||||
])),
|
||||
)
|
||||
.await?;
|
||||
|
||||
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async fn split_by_state(
|
||||
client: &Qdrant,
|
||||
latest_chunks: &[Chunk],
|
||||
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
|
||||
let incoming: HashMap<String, Chunk> = latest_chunks
|
||||
.iter()
|
||||
.map(|c| (c.point_id.clone(), c.clone()))
|
||||
.collect();
|
||||
|
||||
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
|
||||
let points = client
|
||||
.get_points(
|
||||
GetPointsBuilder::new(COLLECTION, ids)
|
||||
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
|
||||
.with_vectors(false),
|
||||
)
|
||||
.await?;
|
||||
|
||||
let mut stored: HashMap<String, String> = HashMap::new();
|
||||
for p in points.result {
|
||||
let hash = p.get("content_hash").as_str().cloned();
|
||||
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
|
||||
(p.id.and_then(|i| i.point_id_options), hash)
|
||||
{
|
||||
stored.insert(id, hash);
|
||||
}
|
||||
}
|
||||
|
||||
let (mut unchanged, mut content_changed, mut unknown_ids) =
|
||||
(Vec::new(), Vec::new(), Vec::new());
|
||||
for (pid, c) in &incoming {
|
||||
if stored.get(pid) == Some(&c.content_hash) {
|
||||
unchanged.push(c.clone());
|
||||
} else if stored.contains_key(pid) {
|
||||
content_changed.push(c.clone());
|
||||
} else {
|
||||
unknown_ids.push(c.clone());
|
||||
}
|
||||
}
|
||||
|
||||
Ok((incoming, unchanged, content_changed, unknown_ids))
|
||||
}
|
||||
|
||||
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||
split_by_state(&client, &latest_chunks).await?;
|
||||
|
||||
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
|
||||
if content_changed.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let points: Vec<PointStruct> = content_changed
|
||||
.iter()
|
||||
.map(|c| {
|
||||
Ok(PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
))
|
||||
})
|
||||
.collect::<anyhow::Result<_>>()?;
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
|
||||
let (mut reused, mut added) = (0, 0);
|
||||
|
||||
for c in unknown_ids {
|
||||
let same_text =
|
||||
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
|
||||
let hits = client
|
||||
.scroll(
|
||||
ScrollPointsBuilder::new(COLLECTION)
|
||||
.filter(same_text)
|
||||
.limit(1)
|
||||
.with_payload(PayloadIncludeSelector::new(vec![
|
||||
"last_updated".to_string()
|
||||
]))
|
||||
.with_vectors(true),
|
||||
)
|
||||
.await?
|
||||
.result;
|
||||
|
||||
let point = if let Some(hit) = hits.into_iter().next() {
|
||||
// same text, new address: copy the vector, keep its last_updated
|
||||
let last_updated = hit.get("last_updated").as_str().cloned();
|
||||
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
|
||||
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
|
||||
Some(vector_output::Vector::Dense(dense)) => dense.data,
|
||||
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||
},
|
||||
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||
};
|
||||
reused += 1;
|
||||
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
|
||||
} else {
|
||||
// genuinely new content: embed and insert
|
||||
added += 1;
|
||||
PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
)
|
||||
};
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
|
||||
.await?;
|
||||
}
|
||||
|
||||
Ok((reused, added))
|
||||
}
|
||||
|
||||
/// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async fn delete_gone(
|
||||
client: &Qdrant,
|
||||
incoming_ids: &HashMap<String, Chunk>,
|
||||
) -> anyhow::Result<u64> {
|
||||
if incoming_ids.is_empty() {
|
||||
anyhow::bail!("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
let stale = Filter::must_not([Condition::has_id(
|
||||
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
|
||||
)]);
|
||||
|
||||
let to_delete = client
|
||||
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
|
||||
.await?
|
||||
.result
|
||||
.map(|r| r.count)
|
||||
.unwrap_or(0);
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client
|
||||
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
|
||||
.await?;
|
||||
Ok(to_delete)
|
||||
}
|
||||
|
||||
async fn sync(
|
||||
client: &Qdrant,
|
||||
latest_chunks: &[Chunk],
|
||||
) -> anyhow::Result<HashMap<&'static str, usize>> {
|
||||
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
|
||||
|
||||
let chunks = prepare_chunks_for_sync(latest_chunks);
|
||||
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||
split_by_state(client, &chunks).await?;
|
||||
|
||||
re_embed_changed(client, &content_changed).await?;
|
||||
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
|
||||
let deleted = delete_gone(client, &incoming_ids).await?;
|
||||
|
||||
Ok(HashMap::from([
|
||||
("unchanged", unchanged.len()),
|
||||
("re-embedded", content_changed.len()),
|
||||
("reused_embedding", reused),
|
||||
("added", added),
|
||||
("deleted", deleted as usize),
|
||||
]))
|
||||
}
|
||||
|
||||
let run = sync(&client, &latest_chunks).await?;
|
||||
println!("{run:?}");
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```csharp
|
||||
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
await client.QueryAsync(
|
||||
collectionName: COLLECTION,
|
||||
query: new Document { Text = QUERY, Model = MODEL },
|
||||
limit: 3,
|
||||
payloadSelector: new[] { "section_url", "text" }
|
||||
);
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```go
|
||||
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||
|
||||
client.Query(context.Background(), &qdrant.QueryPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
|
||||
Limit: qdrant.PtrOf(uint64(3)),
|
||||
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
|
||||
})
|
||||
```
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
```java
|
||||
static final String QUERY =
|
||||
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
static void search() throws Exception {
|
||||
client.queryAsync(
|
||||
QueryPoints.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setQuery(
|
||||
nearest(
|
||||
Document.newBuilder()
|
||||
.setText(QUERY)
|
||||
.setModel(MODEL)
|
||||
.build()))
|
||||
.setLimit(3)
|
||||
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
|
||||
.build()).get();
|
||||
}
|
||||
```
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
```python
|
||||
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||
|
||||
client.query_points(
|
||||
COLLECTION,
|
||||
query=models.Document(text=QUERY, model=MODEL),
|
||||
limit=3,
|
||||
with_payload=["section_url", "text"],
|
||||
)
|
||||
```
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
```rust
|
||||
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
client
|
||||
.query(
|
||||
QueryPointsBuilder::new(COLLECTION)
|
||||
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
|
||||
.limit(3)
|
||||
.with_payload(PayloadIncludeSelector::new(vec![
|
||||
"section_url".to_string(),
|
||||
"text".to_string(),
|
||||
])),
|
||||
)
|
||||
.await?;
|
||||
```
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
```typescript
|
||||
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
await client.query(COLLECTION, {
|
||||
query: { text: QUERY, model: MODEL },
|
||||
limit: 3,
|
||||
with_payload: ["section_url", "text"],
|
||||
});
|
||||
```
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
```csharp
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
|
||||
SplitByState(List<Chunk> latestChunks)
|
||||
{
|
||||
var incoming = latestChunks.ToDictionary(c => c.PointId);
|
||||
|
||||
var stored = new Dictionary<string, string>();
|
||||
var points = await client.RetrieveAsync(
|
||||
COLLECTION,
|
||||
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
|
||||
payloadSelector: new[] { "content_hash" },
|
||||
vectorSelector: false
|
||||
);
|
||||
foreach (var p in points)
|
||||
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
|
||||
|
||||
var unchanged = new List<Chunk>();
|
||||
var contentChanged = new List<Chunk>();
|
||||
var unknownIds = new List<Chunk>();
|
||||
foreach (var (pid, c) in incoming)
|
||||
{
|
||||
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
|
||||
unchanged.Add(c);
|
||||
else if (stored.ContainsKey(pid))
|
||||
contentChanged.Add(c);
|
||||
else
|
||||
unknownIds.Add(c);
|
||||
}
|
||||
|
||||
return (incoming, unchanged, contentChanged, unknownIds);
|
||||
}
|
||||
|
||||
var splitState = await SplitByState(LATEST_CHUNKS);
|
||||
```
|
||||
+39
@@ -0,0 +1,39 @@
|
||||
```go
|
||||
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
|
||||
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
|
||||
incoming := make(map[string]Chunk, len(latestChunks))
|
||||
ids := make([]*qdrant.PointId, 0, len(latestChunks))
|
||||
for _, c := range latestChunks {
|
||||
incoming[c.PointID] = c
|
||||
ids = append(ids, qdrant.NewID(c.PointID))
|
||||
}
|
||||
|
||||
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Ids: ids,
|
||||
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
|
||||
WithVectors: qdrant.NewWithVectors(false),
|
||||
})
|
||||
stored := make(map[string]string, len(retrieved))
|
||||
for _, p := range retrieved {
|
||||
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
|
||||
}
|
||||
|
||||
var unchanged, contentChanged, unknownIDs []Chunk
|
||||
for pid, c := range incoming {
|
||||
storedHash, found := stored[pid]
|
||||
switch {
|
||||
case found && storedHash == c.ContentHash:
|
||||
unchanged = append(unchanged, c)
|
||||
case found:
|
||||
contentChanged = append(contentChanged, c)
|
||||
default:
|
||||
unknownIDs = append(unknownIDs, c)
|
||||
}
|
||||
}
|
||||
|
||||
return incoming, unchanged, contentChanged, unknownIDs
|
||||
}
|
||||
|
||||
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
|
||||
```
|
||||
+43
@@ -0,0 +1,43 @@
|
||||
```java
|
||||
static class SyncState {
|
||||
Map<String, Chunk> incoming = new LinkedHashMap<>();
|
||||
List<Chunk> unchanged = new ArrayList<>();
|
||||
List<Chunk> contentChanged = new ArrayList<>();
|
||||
List<Chunk> unknownIds = new ArrayList<>();
|
||||
}
|
||||
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
|
||||
SyncState state = new SyncState();
|
||||
for (Chunk c : latestChunks) {
|
||||
state.incoming.put(c.pointId, c);
|
||||
}
|
||||
|
||||
Map<String, String> stored = new HashMap<>();
|
||||
var points = client.retrieveAsync(
|
||||
COLLECTION,
|
||||
state.incoming.keySet().stream()
|
||||
.map(pid -> id(UUID.fromString(pid)))
|
||||
.collect(Collectors.toList()),
|
||||
WithPayloadSelectorFactory.include(List.of("content_hash")),
|
||||
WithVectorsSelectorFactory.enable(false),
|
||||
null).get();
|
||||
for (var p : points) {
|
||||
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
|
||||
}
|
||||
|
||||
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
|
||||
String pid = e.getKey();
|
||||
Chunk c = e.getValue();
|
||||
if (c.contentHash.equals(stored.get(pid))) {
|
||||
state.unchanged.add(c);
|
||||
} else if (stored.containsKey(pid)) {
|
||||
state.contentChanged.add(c);
|
||||
} else {
|
||||
state.unknownIds.add(c);
|
||||
}
|
||||
}
|
||||
|
||||
return state;
|
||||
}
|
||||
```
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
```python
|
||||
def split_by_state(latest_chunks):
|
||||
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
||||
incoming = {c["point_id"]: c for c in latest_chunks}
|
||||
|
||||
stored = {}
|
||||
points = client.retrieve(
|
||||
COLLECTION,
|
||||
ids=list(incoming),
|
||||
with_payload=["content_hash"],
|
||||
with_vectors=False,
|
||||
)
|
||||
for p in points:
|
||||
stored[str(p.id)] = p.payload["content_hash"]
|
||||
|
||||
unchanged, content_changed, unknown_ids = [], [], []
|
||||
for pid, c in incoming.items():
|
||||
if stored.get(pid) == c["content_hash"]:
|
||||
unchanged.append(c)
|
||||
elif pid in stored:
|
||||
content_changed.append(c)
|
||||
else:
|
||||
unknown_ids.append(c)
|
||||
|
||||
return incoming, unchanged, content_changed, unknown_ids
|
||||
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
||||
```
|
||||
+48
@@ -0,0 +1,48 @@
|
||||
```rust
|
||||
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async fn split_by_state(
|
||||
client: &Qdrant,
|
||||
latest_chunks: &[Chunk],
|
||||
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
|
||||
let incoming: HashMap<String, Chunk> = latest_chunks
|
||||
.iter()
|
||||
.map(|c| (c.point_id.clone(), c.clone()))
|
||||
.collect();
|
||||
|
||||
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
|
||||
let points = client
|
||||
.get_points(
|
||||
GetPointsBuilder::new(COLLECTION, ids)
|
||||
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
|
||||
.with_vectors(false),
|
||||
)
|
||||
.await?;
|
||||
|
||||
let mut stored: HashMap<String, String> = HashMap::new();
|
||||
for p in points.result {
|
||||
let hash = p.get("content_hash").as_str().cloned();
|
||||
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
|
||||
(p.id.and_then(|i| i.point_id_options), hash)
|
||||
{
|
||||
stored.insert(id, hash);
|
||||
}
|
||||
}
|
||||
|
||||
let (mut unchanged, mut content_changed, mut unknown_ids) =
|
||||
(Vec::new(), Vec::new(), Vec::new());
|
||||
for (pid, c) in &incoming {
|
||||
if stored.get(pid) == Some(&c.content_hash) {
|
||||
unchanged.push(c.clone());
|
||||
} else if stored.contains_key(pid) {
|
||||
content_changed.push(c.clone());
|
||||
} else {
|
||||
unknown_ids.push(c.clone());
|
||||
}
|
||||
}
|
||||
|
||||
Ok((incoming, unchanged, content_changed, unknown_ids))
|
||||
}
|
||||
|
||||
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||
split_by_state(&client, &latest_chunks).await?;
|
||||
```
|
||||
+33
@@ -0,0 +1,33 @@
|
||||
```typescript
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async function splitByState(latestChunks: SyncChunk[]) {
|
||||
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
|
||||
|
||||
const stored = new Map<string, string>();
|
||||
const points = await client.retrieve(COLLECTION, {
|
||||
ids: [...incoming.keys()],
|
||||
with_payload: ["content_hash"],
|
||||
with_vector: false,
|
||||
});
|
||||
for (const p of points) {
|
||||
stored.set(String(p.id), p.payload?.content_hash as string);
|
||||
}
|
||||
|
||||
const unchanged: SyncChunk[] = [];
|
||||
const contentChanged: SyncChunk[] = [];
|
||||
const unknownIds: SyncChunk[] = [];
|
||||
for (const [pid, c] of incoming) {
|
||||
if (stored.get(pid) === c.content_hash) {
|
||||
unchanged.push(c);
|
||||
} else if (stored.has(pid)) {
|
||||
contentChanged.push(c);
|
||||
} else {
|
||||
unknownIds.push(c);
|
||||
}
|
||||
}
|
||||
|
||||
return { incoming, unchanged, contentChanged, unknownIds };
|
||||
}
|
||||
|
||||
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
|
||||
```
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
```csharp
|
||||
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
|
||||
{
|
||||
await CheckGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
var chunks = PrepareChunksForSync(latestChunks);
|
||||
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
|
||||
|
||||
await ReEmbedChanged(contentChanged);
|
||||
var (reused, added) = await ReuseOrAdd(unknownIds);
|
||||
var deleted = await DeleteGone(incomingIds);
|
||||
|
||||
return new Dictionary<string, long>
|
||||
{
|
||||
["unchanged"] = unchanged.Count,
|
||||
["re-embedded"] = contentChanged.Count,
|
||||
["reused_embedding"] = reused,
|
||||
["added"] = added,
|
||||
["deleted"] = (long)deleted,
|
||||
};
|
||||
}
|
||||
```
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
```go
|
||||
sync := func(latestChunks []Chunk) map[string]int {
|
||||
checkGate() // refuse to mix embedding models or pipeline versions
|
||||
|
||||
chunks := prepareChunksForSync(latestChunks)
|
||||
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
|
||||
|
||||
reEmbedChanged(contentChanged)
|
||||
reused, added := reuseOrAdd(unknownIDs)
|
||||
deleted := deleteGone(incomingIDs)
|
||||
|
||||
return map[string]int{
|
||||
"unchanged": len(unchanged),
|
||||
"re-embedded": len(contentChanged),
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
}
|
||||
}
|
||||
```
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
```java
|
||||
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
|
||||
checkGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
List<Chunk> chunks = prepareChunksForSync(latestChunks);
|
||||
SyncState state = splitByState(chunks);
|
||||
|
||||
reEmbedChanged(state.contentChanged);
|
||||
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
|
||||
long deleted = deleteGone(state.incoming);
|
||||
|
||||
return Map.of(
|
||||
"unchanged", (long) state.unchanged.size(),
|
||||
"re-embedded", (long) state.contentChanged.size(),
|
||||
"reused_embedding", (long) reusedAdded[0],
|
||||
"added", (long) reusedAdded[1],
|
||||
"deleted", deleted);
|
||||
}
|
||||
```
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
```python
|
||||
def sync(latest_chunks):
|
||||
check_gate() # refuse to mix embedding models or pipeline versions
|
||||
|
||||
chunks = prepare_chunks_for_sync(latest_chunks)
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
||||
|
||||
re_embed_changed(content_changed)
|
||||
reused, added = reuse_or_add(unknown_ids)
|
||||
deleted = delete_gone(incoming_ids)
|
||||
|
||||
return {
|
||||
"unchanged": len(unchanged),
|
||||
"re-embedded": len(content_changed),
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
}
|
||||
```
|
||||
+24
@@ -0,0 +1,24 @@
|
||||
```rust
|
||||
async fn sync(
|
||||
client: &Qdrant,
|
||||
latest_chunks: &[Chunk],
|
||||
) -> anyhow::Result<HashMap<&'static str, usize>> {
|
||||
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
|
||||
|
||||
let chunks = prepare_chunks_for_sync(latest_chunks);
|
||||
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||
split_by_state(client, &chunks).await?;
|
||||
|
||||
re_embed_changed(client, &content_changed).await?;
|
||||
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
|
||||
let deleted = delete_gone(client, &incoming_ids).await?;
|
||||
|
||||
Ok(HashMap::from([
|
||||
("unchanged", unchanged.len()),
|
||||
("re-embedded", content_changed.len()),
|
||||
("reused_embedding", reused),
|
||||
("added", added),
|
||||
("deleted", deleted as usize),
|
||||
]))
|
||||
}
|
||||
```
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
```typescript
|
||||
async function sync(latestChunks: RawChunk[]) {
|
||||
await checkGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
const chunks = prepareChunksForSync(latestChunks);
|
||||
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
|
||||
|
||||
await reEmbedChanged(contentChanged);
|
||||
const { reused, added } = await reuseOrAdd(unknownIds);
|
||||
const deleted = await deleteGone(incoming);
|
||||
|
||||
return {
|
||||
"unchanged": unchanged.length,
|
||||
"re-embedded": contentChanged.length,
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
};
|
||||
}
|
||||
```
|
||||
+230
@@ -0,0 +1,230 @@
|
||||
```typescript
|
||||
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
|
||||
|
||||
const QDRANT_URL = process.env.QDRANT_URL;
|
||||
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
|
||||
|
||||
const client = new QdrantClient({
|
||||
url: QDRANT_URL,
|
||||
apiKey: QDRANT_API_KEY,
|
||||
});
|
||||
|
||||
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
const PIPELINE = "docs-prep-pipeline-v1";
|
||||
const COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
await client.createCollection(COLLECTION, {
|
||||
vectors: {
|
||||
size: 384, // all-MiniLM-L6-v2 output dimension
|
||||
distance: "Cosine",
|
||||
},
|
||||
});
|
||||
|
||||
await client.updateCollection(COLLECTION, {
|
||||
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
|
||||
});
|
||||
|
||||
async function checkGate() {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
|
||||
{}) as Record<string, unknown>;
|
||||
|
||||
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
|
||||
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
|
||||
}
|
||||
}
|
||||
|
||||
import { createHash } from "node:crypto";
|
||||
|
||||
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
|
||||
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
|
||||
|
||||
function contentHash(text: string): string {
|
||||
return createHash("sha256").update(text).digest("hex");
|
||||
}
|
||||
|
||||
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
|
||||
function pointId(url: string, anchor: string, num: number): string {
|
||||
// Qdrant accepts any well-formed UUID as a point ID:
|
||||
// hash the address, format the digest as a UUID, and the same address always yields the same ID
|
||||
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
|
||||
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
|
||||
}
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
|
||||
return chunks.map((c) => {
|
||||
const text = normalize(c.text);
|
||||
return {
|
||||
...c,
|
||||
text,
|
||||
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
|
||||
content_hash: contentHash(text),
|
||||
point_id: pointId(c.url, c.anchor, c.chunk_num),
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
function payload(chunk: SyncChunk, lastUpdated?: string) {
|
||||
return {
|
||||
url: chunk.url,
|
||||
anchor: chunk.anchor,
|
||||
chunk_num: chunk.chunk_num,
|
||||
section_url: chunk.section_url,
|
||||
text: chunk.text,
|
||||
content_hash: chunk.content_hash,
|
||||
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
|
||||
};
|
||||
}
|
||||
|
||||
for (const field of ["content_hash", "url", "section_url"]) {
|
||||
await client.createPayloadIndex(COLLECTION, {
|
||||
field_name: field,
|
||||
field_schema: "keyword",
|
||||
});
|
||||
}
|
||||
|
||||
await client.upsert(COLLECTION, {
|
||||
points: prepareChunksForSync(CHUNKS).map((c) => ({
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
})),
|
||||
wait: true,
|
||||
});
|
||||
|
||||
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
await client.query(COLLECTION, {
|
||||
query: { text: QUERY, model: MODEL },
|
||||
limit: 3,
|
||||
with_payload: ["section_url", "text"],
|
||||
});
|
||||
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async function splitByState(latestChunks: SyncChunk[]) {
|
||||
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
|
||||
|
||||
const stored = new Map<string, string>();
|
||||
const points = await client.retrieve(COLLECTION, {
|
||||
ids: [...incoming.keys()],
|
||||
with_payload: ["content_hash"],
|
||||
with_vector: false,
|
||||
});
|
||||
for (const p of points) {
|
||||
stored.set(String(p.id), p.payload?.content_hash as string);
|
||||
}
|
||||
|
||||
const unchanged: SyncChunk[] = [];
|
||||
const contentChanged: SyncChunk[] = [];
|
||||
const unknownIds: SyncChunk[] = [];
|
||||
for (const [pid, c] of incoming) {
|
||||
if (stored.get(pid) === c.content_hash) {
|
||||
unchanged.push(c);
|
||||
} else if (stored.has(pid)) {
|
||||
contentChanged.push(c);
|
||||
} else {
|
||||
unknownIds.push(c);
|
||||
}
|
||||
}
|
||||
|
||||
return { incoming, unchanged, contentChanged, unknownIds };
|
||||
}
|
||||
|
||||
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
|
||||
|
||||
async function reEmbedChanged(contentChanged: SyncChunk[]) {
|
||||
if (contentChanged.length === 0) {
|
||||
return;
|
||||
}
|
||||
await client.upsert(COLLECTION, {
|
||||
points: contentChanged.map((c) => ({
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
})),
|
||||
wait: true,
|
||||
});
|
||||
}
|
||||
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async function reuseOrAdd(unknownIds: SyncChunk[]) {
|
||||
let reused = 0;
|
||||
let added = 0;
|
||||
|
||||
for (const c of unknownIds) {
|
||||
const sameText = {
|
||||
must: [
|
||||
{
|
||||
key: "content_hash",
|
||||
match: { value: c.content_hash },
|
||||
},
|
||||
],
|
||||
};
|
||||
const hits = (await client.scroll(COLLECTION, {
|
||||
filter: sameText,
|
||||
limit: 1,
|
||||
with_payload: ["last_updated"],
|
||||
with_vector: true,
|
||||
})).points;
|
||||
|
||||
let point: Schemas["PointStruct"];
|
||||
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
|
||||
point = {
|
||||
id: c.point_id,
|
||||
vector: hits[0].vector as number[],
|
||||
payload: payload(c, hits[0].payload?.last_updated as string),
|
||||
};
|
||||
reused += 1;
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = {
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
};
|
||||
added += 1;
|
||||
}
|
||||
|
||||
await client.upsert(COLLECTION, { points: [point], wait: true });
|
||||
}
|
||||
|
||||
return { reused, added };
|
||||
}
|
||||
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async function deleteGone(incoming: Map<string, SyncChunk>) {
|
||||
if (incoming.size === 0) {
|
||||
throw new Error("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
|
||||
|
||||
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
await client.delete(COLLECTION, { filter: stale, wait: true });
|
||||
return toDelete;
|
||||
}
|
||||
|
||||
async function sync(latestChunks: RawChunk[]) {
|
||||
await checkGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
const chunks = prepareChunksForSync(latestChunks);
|
||||
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
|
||||
|
||||
await reEmbedChanged(contentChanged);
|
||||
const { reused, added } = await reuseOrAdd(unknownIds);
|
||||
const deleted = await deleteGone(incoming);
|
||||
|
||||
return {
|
||||
"unchanged": unchanged.length,
|
||||
"re-embedded": contentChanged.length,
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
};
|
||||
}
|
||||
|
||||
const run = await sync(LATEST_CHUNKS);
|
||||
console.log(run);
|
||||
```
|
||||
+359
@@ -0,0 +1,359 @@
|
||||
package snippet
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"os"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"github.com/qdrant/go-client/qdrant"
|
||||
)
|
||||
|
||||
func Main() {
|
||||
// @block-start client-connection
|
||||
QDRANT_URL := os.Getenv("QDRANT_URL")
|
||||
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
|
||||
|
||||
client, err := qdrant.NewClient(&qdrant.Config{
|
||||
Host: QDRANT_URL,
|
||||
APIKey: QDRANT_API_KEY,
|
||||
UseTLS: true,
|
||||
})
|
||||
// @block-end client-connection
|
||||
|
||||
// @hide-start
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
|
||||
// data and text normalization are not the lesson of this tutorial:
|
||||
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||
type Chunk struct {
|
||||
URL string
|
||||
Anchor string
|
||||
ChunkNum int
|
||||
Text string
|
||||
SectionURL string
|
||||
ContentHash string
|
||||
PointID string
|
||||
}
|
||||
|
||||
CHUNKS := []Chunk{
|
||||
{
|
||||
URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
Anchor: "prerequisites",
|
||||
ChunkNum: 0,
|
||||
Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||
},
|
||||
{
|
||||
URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
Anchor: "step-3-enable-an-admin-api-key",
|
||||
ChunkNum: 0,
|
||||
Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||
},
|
||||
}
|
||||
|
||||
invisibleChars := regexp.MustCompile("[\u200B\u200C\u200D\uFEFF\u00AD]") // zero-width chars and soft hyphen
|
||||
whitespace := regexp.MustCompile(`\s+`)
|
||||
normalize := func(text string) string {
|
||||
text = invisibleChars.ReplaceAllString(text, "")
|
||||
return strings.TrimSpace(whitespace.ReplaceAllString(text, " "))
|
||||
}
|
||||
// @hide-end
|
||||
|
||||
// @block-start create-collection
|
||||
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
|
||||
PIPELINE := "docs-prep-pipeline-v1"
|
||||
COLLECTION := "docs-sync-tutorial"
|
||||
|
||||
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
CollectionName: COLLECTION,
|
||||
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||
Size: 384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance: qdrant.Distance_Cosine,
|
||||
}),
|
||||
Metadata: qdrant.NewValueMap(map[string]any{
|
||||
"embedding_model": MODEL,
|
||||
"pipeline_version": PIPELINE,
|
||||
}),
|
||||
})
|
||||
// @block-end create-collection
|
||||
|
||||
// @block-start check-gate
|
||||
checkGate := func() {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
|
||||
if err != nil { panic(err) } // @hide
|
||||
meta := info.GetConfig().GetMetadata()
|
||||
|
||||
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
|
||||
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
|
||||
}
|
||||
}
|
||||
// @block-end check-gate
|
||||
|
||||
// @block-start identity-and-fingerprint
|
||||
contentHash := func(text string) string {
|
||||
sum := sha256.Sum256([]byte(text))
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
pointID := func(url, anchor string, num int) string {
|
||||
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
|
||||
// marking the input as a URL-like name
|
||||
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
|
||||
}
|
||||
|
||||
// derive both values (and the section address) for every raw chunk
|
||||
prepareChunksForSync := func(chunks []Chunk) []Chunk {
|
||||
out := make([]Chunk, 0, len(chunks))
|
||||
for _, c := range chunks {
|
||||
c.Text = normalize(c.Text)
|
||||
c.SectionURL = c.URL
|
||||
if c.Anchor != "" {
|
||||
c.SectionURL = c.URL + "#" + c.Anchor
|
||||
}
|
||||
c.ContentHash = contentHash(c.Text)
|
||||
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
|
||||
out = append(out, c)
|
||||
}
|
||||
return out
|
||||
}
|
||||
// @block-end identity-and-fingerprint
|
||||
|
||||
// @block-start payload
|
||||
payload := func(c Chunk, lastUpdated string) map[string]any {
|
||||
if lastUpdated == "" {
|
||||
lastUpdated = time.Now().UTC().Format(time.RFC3339)
|
||||
}
|
||||
return map[string]any{
|
||||
"url": c.URL,
|
||||
"anchor": c.Anchor,
|
||||
"chunk_num": c.ChunkNum,
|
||||
"section_url": c.SectionURL,
|
||||
"text": c.Text,
|
||||
"content_hash": c.ContentHash,
|
||||
"last_updated": lastUpdated,
|
||||
}
|
||||
}
|
||||
// @block-end payload
|
||||
|
||||
// @block-start payload-indexes
|
||||
for _, field := range []string{"content_hash", "url", "section_url"} {
|
||||
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
|
||||
CollectionName: COLLECTION,
|
||||
FieldName: field,
|
||||
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
|
||||
})
|
||||
}
|
||||
// @block-end payload-indexes
|
||||
|
||||
// @block-start populate
|
||||
var points []*qdrant.PointStruct
|
||||
for _, c := range prepareChunksForSync(CHUNKS) {
|
||||
points = append(points, &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
})
|
||||
}
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: points,
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
// @block-end populate
|
||||
|
||||
// @block-start search
|
||||
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||
|
||||
client.Query(context.Background(), &qdrant.QueryPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
|
||||
Limit: qdrant.PtrOf(uint64(3)),
|
||||
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
|
||||
})
|
||||
// @block-end search
|
||||
|
||||
// @hide-start
|
||||
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||
LATEST_CHUNKS := prepareChunksForSync(CHUNKS)
|
||||
// @hide-end
|
||||
|
||||
// @block-start split-by-state
|
||||
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
|
||||
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
|
||||
incoming := make(map[string]Chunk, len(latestChunks))
|
||||
ids := make([]*qdrant.PointId, 0, len(latestChunks))
|
||||
for _, c := range latestChunks {
|
||||
incoming[c.PointID] = c
|
||||
ids = append(ids, qdrant.NewID(c.PointID))
|
||||
}
|
||||
|
||||
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Ids: ids,
|
||||
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
|
||||
WithVectors: qdrant.NewWithVectors(false),
|
||||
})
|
||||
if err != nil { panic(err) } // @hide
|
||||
stored := make(map[string]string, len(retrieved))
|
||||
for _, p := range retrieved {
|
||||
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
|
||||
}
|
||||
|
||||
var unchanged, contentChanged, unknownIDs []Chunk
|
||||
for pid, c := range incoming {
|
||||
storedHash, found := stored[pid]
|
||||
switch {
|
||||
case found && storedHash == c.ContentHash:
|
||||
unchanged = append(unchanged, c)
|
||||
case found:
|
||||
contentChanged = append(contentChanged, c)
|
||||
default:
|
||||
unknownIDs = append(unknownIDs, c)
|
||||
}
|
||||
}
|
||||
|
||||
return incoming, unchanged, contentChanged, unknownIDs
|
||||
}
|
||||
|
||||
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
|
||||
// @block-end split-by-state
|
||||
|
||||
// @hide-start
|
||||
_, _, _, _ = incomingIDs, unchanged, contentChanged, unknownIDs
|
||||
// @hide-end
|
||||
|
||||
// @block-start re-embed-changed
|
||||
reEmbedChanged := func(contentChanged []Chunk) {
|
||||
if len(contentChanged) == 0 {
|
||||
return
|
||||
}
|
||||
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
|
||||
for _, c := range contentChanged {
|
||||
points = append(points, &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
})
|
||||
}
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: points,
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
}
|
||||
// @block-end re-embed-changed
|
||||
|
||||
// @block-start reuse-or-add
|
||||
// reuse an existing embedding when the same text is already stored; embed only what is new
|
||||
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
|
||||
reused, added := 0, 0
|
||||
|
||||
for _, c := range unknownIDs {
|
||||
sameText := &qdrant.Filter{
|
||||
Must: []*qdrant.Condition{
|
||||
qdrant.NewMatch("content_hash", c.ContentHash),
|
||||
},
|
||||
}
|
||||
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Filter: sameText,
|
||||
Limit: qdrant.PtrOf(uint32(1)),
|
||||
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
|
||||
WithVectors: qdrant.NewWithVectors(true),
|
||||
})
|
||||
if err != nil { panic(err) } // @hide
|
||||
|
||||
var point *qdrant.PointStruct
|
||||
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
|
||||
point = &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
|
||||
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
|
||||
}
|
||||
reused++
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = &qdrant.PointStruct{
|
||||
Id: qdrant.NewID(c.PointID),
|
||||
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||
}
|
||||
added++
|
||||
}
|
||||
|
||||
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: []*qdrant.PointStruct{point},
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
}
|
||||
|
||||
return reused, added
|
||||
}
|
||||
// @block-end reuse-or-add
|
||||
|
||||
// @block-start delete-gone
|
||||
// remove every point the current crawl no longer contains, return how many
|
||||
deleteGone := func(incomingIDs map[string]Chunk) int {
|
||||
if len(incomingIDs) == 0 {
|
||||
panic("Refusing to delete from an empty source snapshot.")
|
||||
}
|
||||
|
||||
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
|
||||
for pid := range incomingIDs {
|
||||
ids = append(ids, qdrant.NewID(pid))
|
||||
}
|
||||
stale := &qdrant.Filter{
|
||||
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
|
||||
}
|
||||
|
||||
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
|
||||
CollectionName: COLLECTION,
|
||||
Filter: stale,
|
||||
})
|
||||
if err != nil { panic(err) } // @hide
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.Delete(context.Background(), &qdrant.DeletePoints{
|
||||
CollectionName: COLLECTION,
|
||||
Points: qdrant.NewPointsSelectorFilter(stale),
|
||||
Wait: qdrant.PtrOf(true),
|
||||
})
|
||||
return int(toDelete)
|
||||
}
|
||||
// @block-end delete-gone
|
||||
|
||||
// @block-start sync
|
||||
sync := func(latestChunks []Chunk) map[string]int {
|
||||
checkGate() // refuse to mix embedding models or pipeline versions
|
||||
|
||||
chunks := prepareChunksForSync(latestChunks)
|
||||
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
|
||||
|
||||
reEmbedChanged(contentChanged)
|
||||
reused, added := reuseOrAdd(unknownIDs)
|
||||
deleted := deleteGone(incomingIDs)
|
||||
|
||||
return map[string]int{
|
||||
"unchanged": len(unchanged),
|
||||
"re-embedded": len(contentChanged),
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
}
|
||||
}
|
||||
// @block-end sync
|
||||
|
||||
// @block-start run-sync
|
||||
run := sync(LATEST_CHUNKS)
|
||||
fmt.Println(run)
|
||||
// @block-end run-sync
|
||||
}
|
||||
+415
@@ -0,0 +1,415 @@
|
||||
package com.example.snippets_amalgamation;
|
||||
|
||||
import static io.qdrant.client.ConditionFactory.hasId;
|
||||
import static io.qdrant.client.ConditionFactory.matchKeyword;
|
||||
import static io.qdrant.client.PointIdFactory.id;
|
||||
import static io.qdrant.client.QueryFactory.nearest;
|
||||
import static io.qdrant.client.ValueFactory.value;
|
||||
import static io.qdrant.client.VectorFactory.vector;
|
||||
import static io.qdrant.client.VectorsFactory.vectors;
|
||||
|
||||
import io.qdrant.client.QdrantClient;
|
||||
import io.qdrant.client.QdrantGrpcClient;
|
||||
import io.qdrant.client.VectorOutputHelper;
|
||||
import io.qdrant.client.WithPayloadSelectorFactory;
|
||||
import io.qdrant.client.WithVectorsSelectorFactory;
|
||||
import io.qdrant.client.grpc.Collections.CreateCollection;
|
||||
import io.qdrant.client.grpc.Collections.Distance;
|
||||
import io.qdrant.client.grpc.Collections.PayloadSchemaType;
|
||||
import io.qdrant.client.grpc.Collections.VectorParams;
|
||||
import io.qdrant.client.grpc.Collections.VectorsConfig;
|
||||
import io.qdrant.client.grpc.Common.Filter;
|
||||
import io.qdrant.client.grpc.JsonWithInt.Value;
|
||||
import io.qdrant.client.grpc.Points.Document;
|
||||
import io.qdrant.client.grpc.Points.PointStruct;
|
||||
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||
import io.qdrant.client.grpc.Points.ScrollPoints;
|
||||
import java.math.BigInteger;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.time.OffsetDateTime;
|
||||
import java.time.ZoneOffset;
|
||||
import java.time.temporal.ChronoUnit;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.UUID;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
public class Snippet {
|
||||
|
||||
// @block-start client-connection
|
||||
static final String QDRANT_URL = System.getenv("QDRANT_URL");
|
||||
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
|
||||
|
||||
static final QdrantClient client =
|
||||
new QdrantClient(
|
||||
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
|
||||
.withApiKey(QDRANT_API_KEY)
|
||||
.build());
|
||||
// @block-end client-connection
|
||||
|
||||
// @hide-start
|
||||
// data and text normalization are not the lesson of this tutorial:
|
||||
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||
static class Chunk {
|
||||
String url;
|
||||
String anchor;
|
||||
int chunkNum;
|
||||
String text;
|
||||
String sectionUrl; // derived in prepareChunksForSync
|
||||
String contentHash; // derived in prepareChunksForSync
|
||||
String pointId; // derived in prepareChunksForSync
|
||||
|
||||
Chunk(String url, String anchor, int chunkNum, String text) {
|
||||
this.url = url;
|
||||
this.anchor = anchor;
|
||||
this.chunkNum = chunkNum;
|
||||
this.text = text;
|
||||
}
|
||||
}
|
||||
|
||||
static final List<Chunk> CHUNKS = List.of(
|
||||
new Chunk(
|
||||
"https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
"prerequisites",
|
||||
0,
|
||||
"Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ..."),
|
||||
new Chunk(
|
||||
"https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
"step-3-enable-an-admin-api-key",
|
||||
0,
|
||||
"Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ..."));
|
||||
|
||||
static String normalize(String text) {
|
||||
return text.replaceAll("\\s+", " ").strip();
|
||||
}
|
||||
// @hide-end
|
||||
|
||||
// @block-start create-collection
|
||||
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
static final String PIPELINE = "docs-prep-pipeline-v1";
|
||||
static final String COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
static void createCollection() throws Exception {
|
||||
client.createCollectionAsync(
|
||||
CreateCollection.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setVectorsConfig(
|
||||
VectorsConfig.newBuilder()
|
||||
.setParams(
|
||||
VectorParams.newBuilder()
|
||||
.setSize(384) // all-MiniLM-L6-v2 output dimension
|
||||
.setDistance(Distance.Cosine)
|
||||
.build())
|
||||
.build())
|
||||
.putAllMetadata(
|
||||
Map.of(
|
||||
"embedding_model", value(MODEL),
|
||||
"pipeline_version", value(PIPELINE)))
|
||||
.build()).get();
|
||||
}
|
||||
// @block-end create-collection
|
||||
|
||||
// @block-start check-gate
|
||||
static void checkGate() throws Exception {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
Map<String, Value> meta =
|
||||
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
|
||||
|
||||
Value model = meta.get("embedding_model");
|
||||
Value pipeline = meta.get("pipeline_version");
|
||||
if (model == null || !MODEL.equals(model.getStringValue())
|
||||
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
|
||||
throw new RuntimeException(
|
||||
"collection was built by " + meta + ": full re-embed into a fresh collection required");
|
||||
}
|
||||
}
|
||||
// @block-end check-gate
|
||||
|
||||
// @block-start identity-and-fingerprint
|
||||
static String contentHash(String text) throws Exception {
|
||||
byte[] digest = MessageDigest.getInstance("SHA-256")
|
||||
.digest(text.getBytes(StandardCharsets.UTF_8));
|
||||
return String.format("%064x", new BigInteger(1, digest));
|
||||
}
|
||||
|
||||
static String pointId(String url, String anchor, int num) {
|
||||
// name-based UUID (version 3); the same address always yields the same ID
|
||||
return UUID.nameUUIDFromBytes(
|
||||
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
|
||||
}
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
|
||||
List<Chunk> out = new ArrayList<>();
|
||||
for (Chunk c : chunks) {
|
||||
String text = normalize(c.text);
|
||||
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
|
||||
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
|
||||
prepared.contentHash = contentHash(text);
|
||||
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
|
||||
out.add(prepared);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
// @block-end identity-and-fingerprint
|
||||
|
||||
// @block-start payload
|
||||
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
|
||||
Map<String, Value> p = new HashMap<>();
|
||||
p.put("url", value(chunk.url));
|
||||
p.put("anchor", value(chunk.anchor));
|
||||
p.put("chunk_num", value(chunk.chunkNum));
|
||||
p.put("section_url", value(chunk.sectionUrl));
|
||||
p.put("text", value(chunk.text));
|
||||
p.put("content_hash", value(chunk.contentHash));
|
||||
p.put("last_updated", value(lastUpdated != null
|
||||
? lastUpdated
|
||||
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
|
||||
return p;
|
||||
}
|
||||
// @block-end payload
|
||||
|
||||
// @block-start payload-indexes
|
||||
static void createPayloadIndexes() throws Exception {
|
||||
for (String field : List.of("content_hash", "url", "section_url")) {
|
||||
client.createPayloadIndexAsync(
|
||||
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
|
||||
}
|
||||
}
|
||||
// @block-end payload-indexes
|
||||
|
||||
// @block-start populate
|
||||
static void populate() throws Exception {
|
||||
List<PointStruct> points = new ArrayList<>();
|
||||
for (Chunk c : prepareChunksForSync(CHUNKS)) {
|
||||
points.add(
|
||||
PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build());
|
||||
}
|
||||
client.upsertAsync(COLLECTION, points).get();
|
||||
}
|
||||
// @block-end populate
|
||||
|
||||
// @block-start search
|
||||
static final String QUERY =
|
||||
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
static void search() throws Exception {
|
||||
client.queryAsync(
|
||||
QueryPoints.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setQuery(
|
||||
nearest(
|
||||
Document.newBuilder()
|
||||
.setText(QUERY)
|
||||
.setModel(MODEL)
|
||||
.build()))
|
||||
.setLimit(3)
|
||||
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
|
||||
.build()).get();
|
||||
}
|
||||
// @block-end search
|
||||
|
||||
// @hide-start
|
||||
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||
static List<Chunk> LATEST_CHUNKS;
|
||||
// @hide-end
|
||||
|
||||
// @block-start split-by-state
|
||||
static class SyncState {
|
||||
Map<String, Chunk> incoming = new LinkedHashMap<>();
|
||||
List<Chunk> unchanged = new ArrayList<>();
|
||||
List<Chunk> contentChanged = new ArrayList<>();
|
||||
List<Chunk> unknownIds = new ArrayList<>();
|
||||
}
|
||||
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
|
||||
SyncState state = new SyncState();
|
||||
for (Chunk c : latestChunks) {
|
||||
state.incoming.put(c.pointId, c);
|
||||
}
|
||||
|
||||
Map<String, String> stored = new HashMap<>();
|
||||
var points = client.retrieveAsync(
|
||||
COLLECTION,
|
||||
state.incoming.keySet().stream()
|
||||
.map(pid -> id(UUID.fromString(pid)))
|
||||
.collect(Collectors.toList()),
|
||||
WithPayloadSelectorFactory.include(List.of("content_hash")),
|
||||
WithVectorsSelectorFactory.enable(false),
|
||||
null).get();
|
||||
for (var p : points) {
|
||||
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
|
||||
}
|
||||
|
||||
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
|
||||
String pid = e.getKey();
|
||||
Chunk c = e.getValue();
|
||||
if (c.contentHash.equals(stored.get(pid))) {
|
||||
state.unchanged.add(c);
|
||||
} else if (stored.containsKey(pid)) {
|
||||
state.contentChanged.add(c);
|
||||
} else {
|
||||
state.unknownIds.add(c);
|
||||
}
|
||||
}
|
||||
|
||||
return state;
|
||||
}
|
||||
// @block-end split-by-state
|
||||
|
||||
// @block-start re-embed-changed
|
||||
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
|
||||
if (contentChanged.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
List<PointStruct> points = new ArrayList<>();
|
||||
for (Chunk c : contentChanged) {
|
||||
points.add(
|
||||
PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build());
|
||||
}
|
||||
client.upsertAsync(COLLECTION, points).get();
|
||||
}
|
||||
// @block-end re-embed-changed
|
||||
|
||||
// @block-start reuse-or-add
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
|
||||
int reused = 0;
|
||||
int added = 0;
|
||||
|
||||
for (Chunk c : unknownIds) {
|
||||
Filter sameText = Filter.newBuilder()
|
||||
.addMust(matchKeyword("content_hash", c.contentHash))
|
||||
.build();
|
||||
|
||||
var hits = client.scrollAsync(
|
||||
ScrollPoints.newBuilder()
|
||||
.setCollectionName(COLLECTION)
|
||||
.setFilter(sameText)
|
||||
.setLimit(1)
|
||||
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
|
||||
.setWithVectors(WithVectorsSelectorFactory.enable(true))
|
||||
.build()).get().getResultList();
|
||||
|
||||
PointStruct point;
|
||||
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
|
||||
point = PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(vectors(vector(
|
||||
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
|
||||
.getDataList())))
|
||||
.putAllPayload(
|
||||
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
|
||||
.build();
|
||||
reused++;
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = PointStruct.newBuilder()
|
||||
.setId(id(UUID.fromString(c.pointId)))
|
||||
.setVectors(
|
||||
vectors(
|
||||
vector(
|
||||
Document.newBuilder()
|
||||
.setText(c.text)
|
||||
.setModel(MODEL)
|
||||
.build())))
|
||||
.putAllPayload(payload(c, null))
|
||||
.build();
|
||||
added++;
|
||||
}
|
||||
|
||||
client.upsertAsync(COLLECTION, List.of(point)).get();
|
||||
}
|
||||
|
||||
return new int[] {reused, added};
|
||||
}
|
||||
// @block-end reuse-or-add
|
||||
|
||||
// @block-start delete-gone
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
|
||||
if (incomingIds.isEmpty()) {
|
||||
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
Filter stale = Filter.newBuilder()
|
||||
.addMustNot(hasId(
|
||||
incomingIds.keySet().stream()
|
||||
.map(pid -> id(UUID.fromString(pid)))
|
||||
.collect(Collectors.toList())))
|
||||
.build();
|
||||
|
||||
long toDelete = client.countAsync(COLLECTION, stale, true).get();
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.deleteAsync(COLLECTION, stale).get();
|
||||
return toDelete;
|
||||
}
|
||||
// @block-end delete-gone
|
||||
|
||||
// @block-start sync
|
||||
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
|
||||
checkGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
List<Chunk> chunks = prepareChunksForSync(latestChunks);
|
||||
SyncState state = splitByState(chunks);
|
||||
|
||||
reEmbedChanged(state.contentChanged);
|
||||
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
|
||||
long deleted = deleteGone(state.incoming);
|
||||
|
||||
return Map.of(
|
||||
"unchanged", (long) state.unchanged.size(),
|
||||
"re-embedded", (long) state.contentChanged.size(),
|
||||
"reused_embedding", (long) reusedAdded[0],
|
||||
"added", (long) reusedAdded[1],
|
||||
"deleted", deleted);
|
||||
}
|
||||
// @block-end sync
|
||||
|
||||
// @block-start run-sync
|
||||
static void runSync() throws Exception {
|
||||
Map<String, Long> run = sync(LATEST_CHUNKS);
|
||||
System.out.println(run);
|
||||
}
|
||||
// @block-end run-sync
|
||||
|
||||
// @hide-start
|
||||
public static void run() throws Exception {
|
||||
createCollection();
|
||||
createPayloadIndexes();
|
||||
populate();
|
||||
search();
|
||||
|
||||
LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
|
||||
SyncState state = splitByState(LATEST_CHUNKS);
|
||||
|
||||
runSync();
|
||||
// @hide-end
|
||||
}
|
||||
}
|
||||
+261
@@ -0,0 +1,261 @@
|
||||
# @block-start client-connection
|
||||
import os
|
||||
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
QDRANT_URL = os.getenv("QDRANT_URL")
|
||||
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
||||
|
||||
client = QdrantClient(
|
||||
url=QDRANT_URL,
|
||||
api_key=QDRANT_API_KEY,
|
||||
cloud_inference=True
|
||||
)
|
||||
# @block-end client-connection
|
||||
|
||||
# @hide-start
|
||||
# data and text normalization are not the lesson of this tutorial:
|
||||
# the full CHUNKS list and normalize() live in the tutorial notebook
|
||||
CHUNKS = [
|
||||
{
|
||||
"url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
"anchor": "prerequisites",
|
||||
"chunk_num": 0,
|
||||
"text": "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||
},
|
||||
{
|
||||
"url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
"anchor": "step-3-enable-an-admin-api-key",
|
||||
"chunk_num": 0,
|
||||
"text": "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||
},
|
||||
]
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
|
||||
def normalize(text):
|
||||
text = unicodedata.normalize("NFKC", text)
|
||||
text = text.translate(dict.fromkeys(map(ord, "")))
|
||||
return re.sub(r"\s+", " ", text).strip()
|
||||
# @hide-end
|
||||
|
||||
# @block-start create-collection
|
||||
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
||||
PIPELINE = "docs-prep-pipeline-v1"
|
||||
COLLECTION = "docs-sync-tutorial"
|
||||
|
||||
client.create_collection(
|
||||
COLLECTION,
|
||||
vectors_config=models.VectorParams(
|
||||
size=384, # all-MiniLM-L6-v2 output dimension
|
||||
distance=models.Distance.COSINE,
|
||||
),
|
||||
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
||||
)
|
||||
# @block-end create-collection
|
||||
|
||||
# @block-start check-gate
|
||||
def check_gate():
|
||||
# compare this pipeline's constants against what the collection records about itself
|
||||
meta = client.get_collection(COLLECTION).config.metadata or {}
|
||||
|
||||
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
||||
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
||||
# @block-end check-gate
|
||||
|
||||
# @block-start identity-and-fingerprint
|
||||
import hashlib
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
|
||||
def content_hash(text):
|
||||
return hashlib.sha256(text.encode()).hexdigest()
|
||||
|
||||
def point_id(url, anchor, num):
|
||||
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
||||
|
||||
def prepare_chunks_for_sync(chunks):
|
||||
"""Derive both values (and the section address) for every raw chunk."""
|
||||
out = []
|
||||
for c in chunks:
|
||||
text = normalize(c["text"])
|
||||
out.append({
|
||||
**c,
|
||||
"text": text,
|
||||
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
||||
"content_hash": content_hash(text),
|
||||
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
||||
})
|
||||
return out
|
||||
# @block-end identity-and-fingerprint
|
||||
|
||||
# @block-start payload
|
||||
def payload(chunk, last_updated=None):
|
||||
return {
|
||||
"url": chunk["url"],
|
||||
"anchor": chunk["anchor"],
|
||||
"chunk_num": chunk["chunk_num"],
|
||||
"section_url": chunk["section_url"],
|
||||
"text": chunk["text"],
|
||||
"content_hash": chunk["content_hash"],
|
||||
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||
}
|
||||
# @block-end payload
|
||||
|
||||
# @block-start payload-indexes
|
||||
for field in ("content_hash", "url", "section_url"):
|
||||
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
||||
# @block-end payload-indexes
|
||||
|
||||
# @block-start populate
|
||||
client.upsert(COLLECTION, points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in prepare_chunks_for_sync(CHUNKS)
|
||||
], wait=True)
|
||||
# @block-end populate
|
||||
|
||||
# @block-start search
|
||||
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||
|
||||
client.query_points(
|
||||
COLLECTION,
|
||||
query=models.Document(text=QUERY, model=MODEL),
|
||||
limit=3,
|
||||
with_payload=["section_url", "text"],
|
||||
)
|
||||
# @block-end search
|
||||
|
||||
# @hide-start
|
||||
# the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||
LATEST_CHUNKS = prepare_chunks_for_sync(CHUNKS)
|
||||
# @hide-end
|
||||
|
||||
# @block-start split-by-state
|
||||
def split_by_state(latest_chunks):
|
||||
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
||||
incoming = {c["point_id"]: c for c in latest_chunks}
|
||||
|
||||
stored = {}
|
||||
points = client.retrieve(
|
||||
COLLECTION,
|
||||
ids=list(incoming),
|
||||
with_payload=["content_hash"],
|
||||
with_vectors=False,
|
||||
)
|
||||
for p in points:
|
||||
stored[str(p.id)] = p.payload["content_hash"]
|
||||
|
||||
unchanged, content_changed, unknown_ids = [], [], []
|
||||
for pid, c in incoming.items():
|
||||
if stored.get(pid) == c["content_hash"]:
|
||||
unchanged.append(c)
|
||||
elif pid in stored:
|
||||
content_changed.append(c)
|
||||
else:
|
||||
unknown_ids.append(c)
|
||||
|
||||
return incoming, unchanged, content_changed, unknown_ids
|
||||
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
||||
# @block-end split-by-state
|
||||
|
||||
# @block-start re-embed-changed
|
||||
def re_embed_changed(content_changed):
|
||||
if not content_changed:
|
||||
return
|
||||
client.upsert(COLLECTION,
|
||||
points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in content_changed],
|
||||
wait=True)
|
||||
# @block-end re-embed-changed
|
||||
|
||||
# @block-start reuse-or-add
|
||||
def reuse_or_add(unknown_ids):
|
||||
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
||||
reused, added = 0, 0
|
||||
|
||||
for c in unknown_ids:
|
||||
same_text = models.Filter(must=[
|
||||
models.FieldCondition(
|
||||
key="content_hash",
|
||||
match=models.MatchValue(value=c["content_hash"]),
|
||||
)
|
||||
])
|
||||
hits, _ = client.scroll(
|
||||
COLLECTION,
|
||||
scroll_filter=same_text,
|
||||
limit=1,
|
||||
with_payload=["last_updated"],
|
||||
with_vectors=True,
|
||||
)
|
||||
|
||||
if hits: # same text, new address: copy the vector, keep its last_updated
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=hits[0].vector,
|
||||
payload=payload(c, hits[0].payload["last_updated"]),
|
||||
)
|
||||
reused += 1
|
||||
else: # genuinely new content: embed and insert
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
added += 1
|
||||
|
||||
client.upsert(COLLECTION, points=[point], wait=True)
|
||||
|
||||
return reused, added
|
||||
# @block-end reuse-or-add
|
||||
|
||||
# @block-start delete-gone
|
||||
def delete_gone(incoming_ids):
|
||||
"""Remove every point the current crawl no longer contains. Returns how many."""
|
||||
if not incoming_ids:
|
||||
raise ValueError("Refusing to delete from an empty source snapshot.")
|
||||
|
||||
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
||||
|
||||
to_delete = client.count(COLLECTION, count_filter=stale).count
|
||||
|
||||
# potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
||||
return to_delete
|
||||
# @block-end delete-gone
|
||||
|
||||
# @block-start sync
|
||||
def sync(latest_chunks):
|
||||
check_gate() # refuse to mix embedding models or pipeline versions
|
||||
|
||||
chunks = prepare_chunks_for_sync(latest_chunks)
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
||||
|
||||
re_embed_changed(content_changed)
|
||||
reused, added = reuse_or_add(unknown_ids)
|
||||
deleted = delete_gone(incoming_ids)
|
||||
|
||||
return {
|
||||
"unchanged": len(unchanged),
|
||||
"re-embedded": len(content_changed),
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
}
|
||||
# @block-end sync
|
||||
|
||||
# @block-start run-sync
|
||||
run = sync(LATEST_CHUNKS)
|
||||
print(run)
|
||||
# @block-end run-sync
|
||||
+397
@@ -0,0 +1,397 @@
|
||||
use serde_json::{json, Value};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use qdrant_client::qdrant::{
|
||||
point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
|
||||
CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
|
||||
Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
|
||||
Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
|
||||
};
|
||||
use qdrant_client::{Payload, Qdrant};
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
pub async fn main() -> anyhow::Result<()> {
|
||||
// @block-start client-connection
|
||||
let qdrant_url = std::env::var("QDRANT_URL")?;
|
||||
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
|
||||
|
||||
let client = Qdrant::from_url(&qdrant_url)
|
||||
.api_key(qdrant_api_key)
|
||||
.build()?;
|
||||
// @block-end client-connection
|
||||
|
||||
// @hide-start
|
||||
// data and text normalization are not the lesson of this tutorial:
|
||||
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||
#[derive(Clone, Default)]
|
||||
struct Chunk {
|
||||
url: String,
|
||||
anchor: String,
|
||||
chunk_num: u32,
|
||||
text: String,
|
||||
section_url: String,
|
||||
content_hash: String,
|
||||
point_id: String,
|
||||
}
|
||||
|
||||
let chunks: Vec<Chunk> = vec![
|
||||
Chunk {
|
||||
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
|
||||
anchor: "prerequisites".into(),
|
||||
chunk_num: 0,
|
||||
text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...".into(),
|
||||
..Default::default()
|
||||
},
|
||||
Chunk {
|
||||
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
|
||||
anchor: "step-3-enable-an-admin-api-key".into(),
|
||||
chunk_num: 0,
|
||||
text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...".into(),
|
||||
..Default::default()
|
||||
},
|
||||
];
|
||||
|
||||
fn normalize(text: &str) -> String {
|
||||
text.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
// @hide-end
|
||||
|
||||
// @block-start create-collection
|
||||
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
const PIPELINE: &str = "docs-prep-pipeline-v1";
|
||||
const COLLECTION: &str = "docs-sync-tutorial";
|
||||
|
||||
let mut metadata: HashMap<String, Value> = HashMap::new();
|
||||
metadata.insert("embedding_model".to_string(), json!(MODEL));
|
||||
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
|
||||
|
||||
client
|
||||
.create_collection(
|
||||
CreateCollectionBuilder::new(COLLECTION)
|
||||
.vectors_config(VectorParamsBuilder::new(
|
||||
384, // all-MiniLM-L6-v2 output dimension
|
||||
Distance::Cosine,
|
||||
))
|
||||
.metadata(metadata),
|
||||
)
|
||||
.await?;
|
||||
// @block-end create-collection
|
||||
|
||||
// @block-start check-gate
|
||||
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
let meta = client
|
||||
.collection_info(COLLECTION)
|
||||
.await?
|
||||
.result
|
||||
.and_then(|info| info.config)
|
||||
.map(|config| config.metadata)
|
||||
.unwrap_or_default();
|
||||
|
||||
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|
||||
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
|
||||
!= Some(PIPELINE)
|
||||
{
|
||||
anyhow::bail!(
|
||||
"collection was built by {meta:?}: full re-embed into a fresh collection required"
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
// @block-end check-gate
|
||||
|
||||
// @block-start identity-and-fingerprint
|
||||
fn content_hash(text: &str) -> String {
|
||||
Sha256::digest(text.as_bytes())
|
||||
.iter()
|
||||
.map(|byte| format!("{byte:02x}"))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn point_id(url: &str, anchor: &str, num: u32) -> String {
|
||||
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||
uuid::Uuid::new_v5(
|
||||
&uuid::Uuid::NAMESPACE_URL,
|
||||
format!("{url}#{anchor}::{num}").as_bytes(),
|
||||
)
|
||||
.to_string()
|
||||
}
|
||||
|
||||
/// Derive both values (and the section address) for every raw chunk.
|
||||
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
|
||||
chunks
|
||||
.iter()
|
||||
.map(|c| {
|
||||
let text = normalize(&c.text);
|
||||
Chunk {
|
||||
text: text.clone(),
|
||||
section_url: if c.anchor.is_empty() {
|
||||
c.url.clone()
|
||||
} else {
|
||||
format!("{}#{}", c.url, c.anchor)
|
||||
},
|
||||
content_hash: content_hash(&text),
|
||||
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
|
||||
..c.clone()
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
// @block-end identity-and-fingerprint
|
||||
|
||||
// @block-start payload
|
||||
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
|
||||
let last_updated = last_updated.unwrap_or_else(|| {
|
||||
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
|
||||
});
|
||||
Ok(Payload::try_from(serde_json::json!({
|
||||
"url": chunk.url,
|
||||
"anchor": chunk.anchor,
|
||||
"chunk_num": chunk.chunk_num,
|
||||
"section_url": chunk.section_url,
|
||||
"text": chunk.text,
|
||||
"content_hash": chunk.content_hash,
|
||||
"last_updated": last_updated,
|
||||
}))?)
|
||||
}
|
||||
// @block-end payload
|
||||
|
||||
// @block-start payload-indexes
|
||||
for field in ["content_hash", "url", "section_url"] {
|
||||
client
|
||||
.create_field_index(CreateFieldIndexCollectionBuilder::new(
|
||||
COLLECTION,
|
||||
field,
|
||||
FieldType::Keyword,
|
||||
))
|
||||
.await?;
|
||||
}
|
||||
// @block-end payload-indexes
|
||||
|
||||
// @block-start populate
|
||||
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
|
||||
.iter()
|
||||
.map(|c| {
|
||||
Ok(PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
))
|
||||
})
|
||||
.collect::<anyhow::Result<_>>()?;
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||
.await?;
|
||||
// @block-end populate
|
||||
|
||||
// @block-start search
|
||||
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
client
|
||||
.query(
|
||||
QueryPointsBuilder::new(COLLECTION)
|
||||
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
|
||||
.limit(3)
|
||||
.with_payload(PayloadIncludeSelector::new(vec![
|
||||
"section_url".to_string(),
|
||||
"text".to_string(),
|
||||
])),
|
||||
)
|
||||
.await?;
|
||||
// @block-end search
|
||||
|
||||
// @hide-start
|
||||
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||
let latest_chunks = prepare_chunks_for_sync(&chunks);
|
||||
// @hide-end
|
||||
|
||||
// @block-start split-by-state
|
||||
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async fn split_by_state(
|
||||
client: &Qdrant,
|
||||
latest_chunks: &[Chunk],
|
||||
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
|
||||
let incoming: HashMap<String, Chunk> = latest_chunks
|
||||
.iter()
|
||||
.map(|c| (c.point_id.clone(), c.clone()))
|
||||
.collect();
|
||||
|
||||
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
|
||||
let points = client
|
||||
.get_points(
|
||||
GetPointsBuilder::new(COLLECTION, ids)
|
||||
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
|
||||
.with_vectors(false),
|
||||
)
|
||||
.await?;
|
||||
|
||||
let mut stored: HashMap<String, String> = HashMap::new();
|
||||
for p in points.result {
|
||||
let hash = p.get("content_hash").as_str().cloned();
|
||||
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
|
||||
(p.id.and_then(|i| i.point_id_options), hash)
|
||||
{
|
||||
stored.insert(id, hash);
|
||||
}
|
||||
}
|
||||
|
||||
let (mut unchanged, mut content_changed, mut unknown_ids) =
|
||||
(Vec::new(), Vec::new(), Vec::new());
|
||||
for (pid, c) in &incoming {
|
||||
if stored.get(pid) == Some(&c.content_hash) {
|
||||
unchanged.push(c.clone());
|
||||
} else if stored.contains_key(pid) {
|
||||
content_changed.push(c.clone());
|
||||
} else {
|
||||
unknown_ids.push(c.clone());
|
||||
}
|
||||
}
|
||||
|
||||
Ok((incoming, unchanged, content_changed, unknown_ids))
|
||||
}
|
||||
|
||||
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||
split_by_state(&client, &latest_chunks).await?;
|
||||
// @block-end split-by-state
|
||||
|
||||
// @hide-start
|
||||
_ = (&incoming_ids, &unchanged, &content_changed, &unknown_ids);
|
||||
// @hide-end
|
||||
|
||||
// @block-start re-embed-changed
|
||||
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
|
||||
if content_changed.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let points: Vec<PointStruct> = content_changed
|
||||
.iter()
|
||||
.map(|c| {
|
||||
Ok(PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
))
|
||||
})
|
||||
.collect::<anyhow::Result<_>>()?;
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
// @block-end re-embed-changed
|
||||
|
||||
// @block-start reuse-or-add
|
||||
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
|
||||
let (mut reused, mut added) = (0, 0);
|
||||
|
||||
for c in unknown_ids {
|
||||
let same_text =
|
||||
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
|
||||
let hits = client
|
||||
.scroll(
|
||||
ScrollPointsBuilder::new(COLLECTION)
|
||||
.filter(same_text)
|
||||
.limit(1)
|
||||
.with_payload(PayloadIncludeSelector::new(vec![
|
||||
"last_updated".to_string()
|
||||
]))
|
||||
.with_vectors(true),
|
||||
)
|
||||
.await?
|
||||
.result;
|
||||
|
||||
let point = if let Some(hit) = hits.into_iter().next() {
|
||||
// same text, new address: copy the vector, keep its last_updated
|
||||
let last_updated = hit.get("last_updated").as_str().cloned();
|
||||
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
|
||||
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
|
||||
Some(vector_output::Vector::Dense(dense)) => dense.data,
|
||||
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||
},
|
||||
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||
};
|
||||
reused += 1;
|
||||
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
|
||||
} else {
|
||||
// genuinely new content: embed and insert
|
||||
added += 1;
|
||||
PointStruct::new(
|
||||
c.point_id.clone(),
|
||||
Document::new(&c.text, MODEL),
|
||||
payload(c, None)?,
|
||||
)
|
||||
};
|
||||
|
||||
client
|
||||
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
|
||||
.await?;
|
||||
}
|
||||
|
||||
Ok((reused, added))
|
||||
}
|
||||
// @block-end reuse-or-add
|
||||
|
||||
// @block-start delete-gone
|
||||
/// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async fn delete_gone(
|
||||
client: &Qdrant,
|
||||
incoming_ids: &HashMap<String, Chunk>,
|
||||
) -> anyhow::Result<u64> {
|
||||
if incoming_ids.is_empty() {
|
||||
anyhow::bail!("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
let stale = Filter::must_not([Condition::has_id(
|
||||
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
|
||||
)]);
|
||||
|
||||
let to_delete = client
|
||||
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
|
||||
.await?
|
||||
.result
|
||||
.map(|r| r.count)
|
||||
.unwrap_or(0);
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client
|
||||
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
|
||||
.await?;
|
||||
Ok(to_delete)
|
||||
}
|
||||
// @block-end delete-gone
|
||||
|
||||
// @block-start sync
|
||||
async fn sync(
|
||||
client: &Qdrant,
|
||||
latest_chunks: &[Chunk],
|
||||
) -> anyhow::Result<HashMap<&'static str, usize>> {
|
||||
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
|
||||
|
||||
let chunks = prepare_chunks_for_sync(latest_chunks);
|
||||
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||
split_by_state(client, &chunks).await?;
|
||||
|
||||
re_embed_changed(client, &content_changed).await?;
|
||||
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
|
||||
let deleted = delete_gone(client, &incoming_ids).await?;
|
||||
|
||||
Ok(HashMap::from([
|
||||
("unchanged", unchanged.len()),
|
||||
("re-embedded", content_changed.len()),
|
||||
("reused_embedding", reused),
|
||||
("added", added),
|
||||
("deleted", deleted as usize),
|
||||
]))
|
||||
}
|
||||
// @block-end sync
|
||||
|
||||
// @block-start run-sync
|
||||
let run = sync(&client, &latest_chunks).await?;
|
||||
println!("{run:?}");
|
||||
// @block-end run-sync
|
||||
|
||||
Ok(())
|
||||
}
|
||||
+288
@@ -0,0 +1,288 @@
|
||||
// @block-start client-connection
|
||||
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
|
||||
|
||||
const QDRANT_URL = process.env.QDRANT_URL;
|
||||
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
|
||||
|
||||
const client = new QdrantClient({
|
||||
url: QDRANT_URL,
|
||||
apiKey: QDRANT_API_KEY,
|
||||
});
|
||||
// @block-end client-connection
|
||||
|
||||
// @hide-start
|
||||
// data and text normalization are not the lesson of this tutorial:
|
||||
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||
const CHUNKS = [
|
||||
{
|
||||
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
anchor: "prerequisites",
|
||||
chunk_num: 0,
|
||||
text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||
},
|
||||
{
|
||||
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||
anchor: "step-3-enable-an-admin-api-key",
|
||||
chunk_num: 0,
|
||||
text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||
},
|
||||
];
|
||||
|
||||
function normalize(text: string): string {
|
||||
return text
|
||||
.normalize("NFKC")
|
||||
.replace(/[\u200B\u200C\u200D\uFEFF\u00AD]/g, "")
|
||||
.replace(/\s+/g, " ")
|
||||
.trim();
|
||||
}
|
||||
// @hide-end
|
||||
|
||||
// @block-start create-collection
|
||||
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||
const PIPELINE = "docs-prep-pipeline-v1";
|
||||
const COLLECTION = "docs-sync-tutorial";
|
||||
|
||||
await client.createCollection(COLLECTION, {
|
||||
vectors: {
|
||||
size: 384, // all-MiniLM-L6-v2 output dimension
|
||||
distance: "Cosine",
|
||||
},
|
||||
});
|
||||
|
||||
await client.updateCollection(COLLECTION, {
|
||||
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
|
||||
});
|
||||
// @block-end create-collection
|
||||
|
||||
// @block-start check-gate
|
||||
async function checkGate() {
|
||||
// compare this pipeline's constants against what the collection records about itself
|
||||
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
|
||||
{}) as Record<string, unknown>;
|
||||
|
||||
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
|
||||
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
|
||||
}
|
||||
}
|
||||
// @block-end check-gate
|
||||
|
||||
// @block-start identity-and-fingerprint
|
||||
import { createHash } from "node:crypto";
|
||||
|
||||
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
|
||||
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
|
||||
|
||||
function contentHash(text: string): string {
|
||||
return createHash("sha256").update(text).digest("hex");
|
||||
}
|
||||
|
||||
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
|
||||
function pointId(url: string, anchor: string, num: number): string {
|
||||
// Qdrant accepts any well-formed UUID as a point ID:
|
||||
// hash the address, format the digest as a UUID, and the same address always yields the same ID
|
||||
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
|
||||
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
|
||||
}
|
||||
|
||||
// Derive both values (and the section address) for every raw chunk.
|
||||
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
|
||||
return chunks.map((c) => {
|
||||
const text = normalize(c.text);
|
||||
return {
|
||||
...c,
|
||||
text,
|
||||
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
|
||||
content_hash: contentHash(text),
|
||||
point_id: pointId(c.url, c.anchor, c.chunk_num),
|
||||
};
|
||||
});
|
||||
}
|
||||
// @block-end identity-and-fingerprint
|
||||
|
||||
// @block-start payload
|
||||
function payload(chunk: SyncChunk, lastUpdated?: string) {
|
||||
return {
|
||||
url: chunk.url,
|
||||
anchor: chunk.anchor,
|
||||
chunk_num: chunk.chunk_num,
|
||||
section_url: chunk.section_url,
|
||||
text: chunk.text,
|
||||
content_hash: chunk.content_hash,
|
||||
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
|
||||
};
|
||||
}
|
||||
// @block-end payload
|
||||
|
||||
// @block-start payload-indexes
|
||||
for (const field of ["content_hash", "url", "section_url"]) {
|
||||
await client.createPayloadIndex(COLLECTION, {
|
||||
field_name: field,
|
||||
field_schema: "keyword",
|
||||
});
|
||||
}
|
||||
// @block-end payload-indexes
|
||||
|
||||
// @block-start populate
|
||||
await client.upsert(COLLECTION, {
|
||||
points: prepareChunksForSync(CHUNKS).map((c) => ({
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
})),
|
||||
wait: true,
|
||||
});
|
||||
// @block-end populate
|
||||
|
||||
// @block-start search
|
||||
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||
|
||||
await client.query(COLLECTION, {
|
||||
query: { text: QUERY, model: MODEL },
|
||||
limit: 3,
|
||||
with_payload: ["section_url", "text"],
|
||||
});
|
||||
// @block-end search
|
||||
|
||||
// @hide-start
|
||||
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||
const LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
|
||||
// @hide-end
|
||||
|
||||
// @block-start split-by-state
|
||||
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||
async function splitByState(latestChunks: SyncChunk[]) {
|
||||
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
|
||||
|
||||
const stored = new Map<string, string>();
|
||||
const points = await client.retrieve(COLLECTION, {
|
||||
ids: [...incoming.keys()],
|
||||
with_payload: ["content_hash"],
|
||||
with_vector: false,
|
||||
});
|
||||
for (const p of points) {
|
||||
stored.set(String(p.id), p.payload?.content_hash as string);
|
||||
}
|
||||
|
||||
const unchanged: SyncChunk[] = [];
|
||||
const contentChanged: SyncChunk[] = [];
|
||||
const unknownIds: SyncChunk[] = [];
|
||||
for (const [pid, c] of incoming) {
|
||||
if (stored.get(pid) === c.content_hash) {
|
||||
unchanged.push(c);
|
||||
} else if (stored.has(pid)) {
|
||||
contentChanged.push(c);
|
||||
} else {
|
||||
unknownIds.push(c);
|
||||
}
|
||||
}
|
||||
|
||||
return { incoming, unchanged, contentChanged, unknownIds };
|
||||
}
|
||||
|
||||
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
|
||||
// @block-end split-by-state
|
||||
|
||||
// @block-start re-embed-changed
|
||||
async function reEmbedChanged(contentChanged: SyncChunk[]) {
|
||||
if (contentChanged.length === 0) {
|
||||
return;
|
||||
}
|
||||
await client.upsert(COLLECTION, {
|
||||
points: contentChanged.map((c) => ({
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
})),
|
||||
wait: true,
|
||||
});
|
||||
}
|
||||
// @block-end re-embed-changed
|
||||
|
||||
// @block-start reuse-or-add
|
||||
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||
async function reuseOrAdd(unknownIds: SyncChunk[]) {
|
||||
let reused = 0;
|
||||
let added = 0;
|
||||
|
||||
for (const c of unknownIds) {
|
||||
const sameText = {
|
||||
must: [
|
||||
{
|
||||
key: "content_hash",
|
||||
match: { value: c.content_hash },
|
||||
},
|
||||
],
|
||||
};
|
||||
const hits = (await client.scroll(COLLECTION, {
|
||||
filter: sameText,
|
||||
limit: 1,
|
||||
with_payload: ["last_updated"],
|
||||
with_vector: true,
|
||||
})).points;
|
||||
|
||||
let point: Schemas["PointStruct"];
|
||||
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
|
||||
point = {
|
||||
id: c.point_id,
|
||||
vector: hits[0].vector as number[],
|
||||
payload: payload(c, hits[0].payload?.last_updated as string),
|
||||
};
|
||||
reused += 1;
|
||||
} else { // genuinely new content: embed and insert
|
||||
point = {
|
||||
id: c.point_id,
|
||||
vector: { text: c.text, model: MODEL },
|
||||
payload: payload(c),
|
||||
};
|
||||
added += 1;
|
||||
}
|
||||
|
||||
await client.upsert(COLLECTION, { points: [point], wait: true });
|
||||
}
|
||||
|
||||
return { reused, added };
|
||||
}
|
||||
// @block-end reuse-or-add
|
||||
|
||||
// @block-start delete-gone
|
||||
// Remove every point the current crawl no longer contains. Returns how many.
|
||||
async function deleteGone(incoming: Map<string, SyncChunk>) {
|
||||
if (incoming.size === 0) {
|
||||
throw new Error("Refusing to delete from an empty source snapshot.");
|
||||
}
|
||||
|
||||
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
|
||||
|
||||
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
|
||||
|
||||
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
await client.delete(COLLECTION, { filter: stale, wait: true });
|
||||
return toDelete;
|
||||
}
|
||||
// @block-end delete-gone
|
||||
|
||||
// @block-start sync
|
||||
async function sync(latestChunks: RawChunk[]) {
|
||||
await checkGate(); // refuse to mix embedding models or pipeline versions
|
||||
|
||||
const chunks = prepareChunksForSync(latestChunks);
|
||||
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
|
||||
|
||||
await reEmbedChanged(contentChanged);
|
||||
const { reused, added } = await reuseOrAdd(unknownIds);
|
||||
const deleted = await deleteGone(incoming);
|
||||
|
||||
return {
|
||||
"unchanged": unchanged.length,
|
||||
"re-embedded": contentChanged.length,
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
};
|
||||
}
|
||||
// @block-end sync
|
||||
|
||||
// @block-start run-sync
|
||||
const run = await sync(LATEST_CHUNKS);
|
||||
console.log(run);
|
||||
// @block-end run-sync
|
||||
+15
-219
@@ -35,27 +35,13 @@ The tutorial has an accompanying [notebook](https://github.com/qdrant/examples/b
|
||||
|
||||
## Prerequisites
|
||||
|
||||
```python
|
||||
%pip install -q "qdrant-client>=1.18"
|
||||
```
|
||||
Install the [Qdrant client of your choice](/documentation/interfaces/#client-libraries).
|
||||
|
||||
We use Qdrant Cloud and its [Free Embedding Inference](/documentation/cloud/inference/#free-embedding-models).
|
||||
Create a Free Tier [Qdrant Cloud cluster](https://cloud.qdrant.io/) and set `QDRANT_URL` and `QDRANT_API_KEY` in your environment.
|
||||
|
||||
|
||||
```python
|
||||
import os
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
QDRANT_URL = os.getenv("QDRANT_URL")
|
||||
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
||||
|
||||
client = QdrantClient(
|
||||
url=QDRANT_URL,
|
||||
api_key=QDRANT_API_KEY,
|
||||
cloud_inference=True
|
||||
)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="client-connection" >}}
|
||||
|
||||
## The Data: Qdrant Documentation
|
||||
|
||||
@@ -133,31 +119,11 @@ Vectors produced by different embedding models, or by the same model over differ
|
||||
|
||||
Let's consider a simple guardrail: save which model and which pipeline version produced the data points, in [**collection metadata**](/documentation/manage-data/collections/#collection-metadata), and verify against it. If one of the two changed, we need to trigger full collection re-embedding.
|
||||
|
||||
```python
|
||||
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
||||
PIPELINE = "docs-prep-pipeline-v1"
|
||||
COLLECTION = "docs-sync-tutorial"
|
||||
|
||||
client.create_collection(
|
||||
COLLECTION,
|
||||
vectors_config=models.VectorParams(
|
||||
size=384, # all-MiniLM-L6-v2 output dimension
|
||||
distance=models.Distance.COSINE,
|
||||
),
|
||||
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
||||
)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="create-collection" >}}
|
||||
|
||||
The gate against mixing embedding generations is then a simple check at the start of every run:
|
||||
|
||||
```python
|
||||
def check_gate():
|
||||
# compare this pipeline's constants against what the collection records about itself
|
||||
meta = client.get_collection(COLLECTION).config.metadata or {}
|
||||
|
||||
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
||||
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="check-gate" >}}
|
||||
|
||||
## Characteristics of a Document Chunk
|
||||
|
||||
@@ -174,32 +140,7 @@ Hence every record should get two derived values:
|
||||
- **Content fingerprint**, like SHA-256 of the text. It changes if a single character changes, and never otherwise. Comparing fingerprints answers "*Is it the same content?*" without comparing texts.
|
||||
- **Deterministic ID** for position in documentation. For example, `url + "#" + anchor + "::" + chunk_num` turned into a UUID, one of the two point ID formats Qdrant accepts. Comparing IDs answers "*Is this content still at the same position?*".
|
||||
|
||||
```python
|
||||
import hashlib
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
|
||||
def content_hash(text):
|
||||
return hashlib.sha256(text.encode()).hexdigest()
|
||||
|
||||
def point_id(url, anchor, num):
|
||||
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
||||
|
||||
def prepare_chunks_for_sync(chunks):
|
||||
"""Derive both values (and the section address) for every raw chunk."""
|
||||
out = []
|
||||
for c in chunks:
|
||||
text = normalize(c["text"])
|
||||
out.append({
|
||||
**c,
|
||||
"text": text,
|
||||
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
||||
"content_hash": content_hash(text),
|
||||
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
||||
})
|
||||
return out
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="identity-and-fingerprint" >}}
|
||||
Example:
|
||||
|
||||
```text
|
||||
@@ -218,56 +159,24 @@ Additionally, a point can be described by the following fields:
|
||||
<details>
|
||||
<summary>payload() implementation</summary>
|
||||
|
||||
```python
|
||||
def payload(chunk, last_updated=None):
|
||||
return {
|
||||
"url": chunk["url"],
|
||||
"anchor": chunk["anchor"],
|
||||
"chunk_num": chunk["chunk_num"],
|
||||
"section_url": chunk["section_url"],
|
||||
"text": chunk["text"],
|
||||
"content_hash": chunk["content_hash"],
|
||||
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||
}
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload" >}}
|
||||
|
||||
</details>
|
||||
|
||||
For all the payload fields used for filtering or grouping we need to create a [**payload index**](/documentation/manage-data/indexing/).
|
||||
|
||||
```python
|
||||
for field in ("content_hash", "url", "section_url"):
|
||||
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload-indexes" >}}
|
||||
|
||||
## Populate Collection
|
||||
|
||||
Populate the collection with the whole documentation.
|
||||
|
||||
```python
|
||||
client.upsert(COLLECTION, points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL), # Cloud Inference embeds text server-side
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in prepare_chunks_for_sync(CHUNKS)
|
||||
], wait=True)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="populate" >}}
|
||||
|
||||
<details>
|
||||
<summary>Test the search against it</summary>
|
||||
|
||||
```python
|
||||
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||
|
||||
client.query_points(
|
||||
COLLECTION,
|
||||
query=models.Document(text=QUERY, model=MODEL),
|
||||
limit=3,
|
||||
with_payload=["section_url", "text"],
|
||||
)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="search" >}}
|
||||
|
||||
You should get something like:
|
||||
|
||||
@@ -354,35 +263,7 @@ We now check every incoming chunk against the collection: does its ID (address)
|
||||
|
||||
[`retrieve`](/documentation/manage-data/points/) fetches points by ID. At corpus scale you would batch the IDs.
|
||||
|
||||
```python
|
||||
def split_by_state(latest_chunks):
|
||||
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
||||
incoming = {c["point_id"]: c for c in latest_chunks}
|
||||
|
||||
stored = {}
|
||||
points = client.retrieve(
|
||||
COLLECTION,
|
||||
ids=list(incoming),
|
||||
with_payload=["content_hash"],
|
||||
with_vectors=False,
|
||||
)
|
||||
for p in points:
|
||||
stored[str(p.id)] = p.payload["content_hash"]
|
||||
|
||||
unchanged, content_changed, unknown_ids = [], [], []
|
||||
for pid, c in incoming.items():
|
||||
if stored.get(pid) == c["content_hash"]:
|
||||
unchanged.append(c)
|
||||
elif pid in stored:
|
||||
content_changed.append(c)
|
||||
else:
|
||||
unknown_ids.append(c)
|
||||
|
||||
return incoming, unchanged, content_changed, unknown_ids
|
||||
|
||||
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="split-by-state" >}}
|
||||
|
||||
### Case 1: Unchanged, Do Nothing
|
||||
|
||||
@@ -393,20 +274,7 @@ These chunks carry the same fingerprint as before.
|
||||
The chunk about Step 3 exists under a known ID (it didn't change its position on the docs website) but carries new information.
|
||||
Use `upsert`: writing a point under an existing ID replaces it.
|
||||
|
||||
```python
|
||||
def re_embed_changed(content_changed):
|
||||
if not content_changed:
|
||||
return
|
||||
client.upsert(COLLECTION,
|
||||
points=[
|
||||
models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
for c in content_changed],
|
||||
wait=True)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="re-embed-changed" >}}
|
||||
|
||||
### Cases 3 and 4: ID Is Not Present in the Collection
|
||||
|
||||
@@ -418,45 +286,7 @@ A filtered [`scroll`](/documentation/manage-data/points/) on `content_hash` answ
|
||||
|
||||
**Note:** *This version performs one hash lookup per unknown chunk so the decision is easy to inspect. In production, batch hash lookups and point upserts.*
|
||||
|
||||
```python
|
||||
def reuse_or_add(unknown_ids):
|
||||
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
||||
reused, added = 0, 0
|
||||
|
||||
for c in unknown_ids:
|
||||
same_text = models.Filter(must=[
|
||||
models.FieldCondition(
|
||||
key="content_hash",
|
||||
match=models.MatchValue(value=c["content_hash"]),
|
||||
)
|
||||
])
|
||||
hits, _ = client.scroll(
|
||||
COLLECTION,
|
||||
scroll_filter=same_text,
|
||||
limit=1,
|
||||
with_payload=["last_updated"],
|
||||
with_vectors=True,
|
||||
)
|
||||
|
||||
if hits: # same text, new address: copy the vector, keep its last_updated
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=hits[0].vector,
|
||||
payload=payload(c, hits[0].payload["last_updated"]),
|
||||
)
|
||||
reused += 1
|
||||
else: # genuinely new content: embed and insert
|
||||
point = models.PointStruct(
|
||||
id=c["point_id"],
|
||||
vector=models.Document(text=c["text"], model=MODEL),
|
||||
payload=payload(c),
|
||||
)
|
||||
added += 1
|
||||
|
||||
client.upsert(COLLECTION, points=[point], wait=True)
|
||||
|
||||
return reused, added
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="reuse-or-add" >}}
|
||||
|
||||
What's important to notice: the old points, the migration page under its old URL, are still in the collection. They need to be removed, and that is the last case.
|
||||
|
||||
@@ -471,51 +301,17 @@ Whatever LATEST_CHUNKS does not contain no longer exists at the source. The dele
|
||||
**Note:** Frequent re-embeddings and deletions don't degrade the index over time: background [optimizers](/documentation/ops-optimization/optimizer/) rebuild and merge index segments as changes accumulate.
|
||||
|
||||
|
||||
```python
|
||||
def delete_gone(incoming_ids):
|
||||
"""Remove every point the current crawl no longer contains. Returns how many."""
|
||||
if not incoming_ids:
|
||||
raise ValueError("Refusing to delete from an empty source snapshot.")
|
||||
|
||||
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
||||
|
||||
to_delete = client.count(COLLECTION, count_filter=stale).count
|
||||
|
||||
# potential check against a threshold to avoid accidental mass deletion could be added here
|
||||
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
||||
return to_delete
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="delete-gone" >}}
|
||||
|
||||
## Run and Verify the Sync
|
||||
|
||||
The five cases, assembled from the functions defined above:
|
||||
|
||||
```python
|
||||
def sync(latest_chunks):
|
||||
check_gate() # refuse to mix embedding models or pipeline versions
|
||||
|
||||
chunks = prepare_chunks_for_sync(latest_chunks)
|
||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
||||
|
||||
re_embed_changed(content_changed)
|
||||
reused, added = reuse_or_add(unknown_ids)
|
||||
deleted = delete_gone(incoming_ids)
|
||||
|
||||
return {
|
||||
"unchanged": len(unchanged),
|
||||
"re-embedded": len(content_changed),
|
||||
"reused_embedding": reused,
|
||||
"added": added,
|
||||
"deleted": deleted,
|
||||
}
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="sync" >}}
|
||||
|
||||
Run the sync.
|
||||
|
||||
```python
|
||||
run = sync(LATEST_CHUNKS)
|
||||
print(run)
|
||||
```
|
||||
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="run-sync" >}}
|
||||
|
||||
You should see something like:
|
||||
|
||||
|
||||
Reference in New Issue
Block a user