mirror of
https://github.com/qdrant/landing_page.git
synced 2026-10-06 19:38:30 +02:00
added snippets in all languages
This commit is contained in:
@@ -15,4 +15,5 @@ serde_json = "1.0.145"
|
|||||||
tempfile = "3"
|
tempfile = "3"
|
||||||
tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] }
|
tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] }
|
||||||
ureq = { version = "3", features = ["json"] }
|
ureq = { version = "3", features = ["json"] }
|
||||||
uuid = { version = "1.18.1", features = ["v4"] }
|
uuid = { version = "1.18.1", features = ["v4", "v5"] }
|
||||||
|
sha2 = "0.11"
|
||||||
|
|||||||
@@ -6,6 +6,6 @@
|
|||||||
| [Time-Based Sharding](/documentation/tutorials-operations/time-based-sharding/) | Efficiently manage time-series data with user-defined sharding. | <span class="pill">Any</span> | 1h | <span class="text-yellow">Intermediate</span> |
|
| [Time-Based Sharding](/documentation/tutorials-operations/time-based-sharding/) | Efficiently manage time-series data with user-defined sharding. | <span class="pill">Any</span> | 1h | <span class="text-yellow">Intermediate</span> |
|
||||||
| [Large-Scale Search](/documentation/tutorials-operations/large-scale-search/) | Cost-efficient search for LAION-400M datasets. | <span class="pill">Any</span> | 48h | <span class="text-red">Advanced</span> |
|
| [Large-Scale Search](/documentation/tutorials-operations/large-scale-search/) | Cost-efficient search for LAION-400M datasets. | <span class="pill">Any</span> | 48h | <span class="text-red">Advanced</span> |
|
||||||
| [Secure a Self-Hosted Instance](/documentation/tutorials-operations/secure-qdrant/) | Enable TLS, API keys, and JWT access control. | <span class="pill">Any</span> | 45m | <span class="text-yellow">Intermediate</span> |
|
| [Secure a Self-Hosted Instance](/documentation/tutorials-operations/secure-qdrant/) | Enable TLS, API keys, and JWT access control. | <span class="pill">Any</span> | 45m | <span class="text-yellow">Intermediate</span> |
|
||||||
| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | <span class="pill">Python</span> | 25m | <span class="text-green">Beginner</span> |
|
| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | <span class="pill">Any</span> | 25m | <span class="text-green">Beginner</span> |
|
||||||
| [Qdrant Cloud Prometheus Monitoring](/documentation/ops-monitoring/managed-cloud-prometheus/) | Observability with Prometheus and Grafana. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
|
| [Qdrant Cloud Prometheus Monitoring](/documentation/ops-monitoring/managed-cloud-prometheus/) | Observability with Prometheus and Grafana. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
|
||||||
| [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
|
| [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | <span class="pill">Prometheus</span> | 30m | <span class="text-yellow">Intermediate</span> |
|
||||||
+310
@@ -0,0 +1,310 @@
|
|||||||
|
using System.Security.Cryptography;
|
||||||
|
using System.Text;
|
||||||
|
using System.Text.RegularExpressions;
|
||||||
|
using Qdrant.Client;
|
||||||
|
using Qdrant.Client.Grpc;
|
||||||
|
using static Qdrant.Client.Grpc.Conditions;
|
||||||
|
using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
|
||||||
|
|
||||||
|
public class Snippet
|
||||||
|
{
|
||||||
|
public static async Task Run()
|
||||||
|
{
|
||||||
|
// @block-start client-connection
|
||||||
|
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
|
||||||
|
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
|
||||||
|
|
||||||
|
var client = new QdrantClient(
|
||||||
|
host: QDRANT_URL!,
|
||||||
|
https: true,
|
||||||
|
apiKey: QDRANT_API_KEY
|
||||||
|
);
|
||||||
|
// @block-end client-connection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// data and text normalization are not the lesson of this tutorial:
|
||||||
|
// the full CHUNKS list and Normalize() live in the tutorial notebook
|
||||||
|
var CHUNKS = new List<Chunk>
|
||||||
|
{
|
||||||
|
(
|
||||||
|
Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
Anchor: "prerequisites",
|
||||||
|
ChunkNum: 0,
|
||||||
|
Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||||
|
SectionUrl: "", ContentHash: "", PointId: ""
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
Anchor: "step-3-enable-an-admin-api-key",
|
||||||
|
ChunkNum: 0,
|
||||||
|
Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||||
|
SectionUrl: "", ContentHash: "", PointId: ""
|
||||||
|
),
|
||||||
|
};
|
||||||
|
|
||||||
|
string Normalize(string text) => Regex.Replace(text, @"\s+", " ").Trim();
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-collection
|
||||||
|
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
var PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
var COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
await client.CreateCollectionAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
vectorsConfig: new VectorParams
|
||||||
|
{
|
||||||
|
Size = 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance = Distance.Cosine
|
||||||
|
},
|
||||||
|
metadata: new()
|
||||||
|
{
|
||||||
|
["embedding_model"] = MODEL,
|
||||||
|
["pipeline_version"] = PIPELINE
|
||||||
|
}
|
||||||
|
);
|
||||||
|
// @block-end create-collection
|
||||||
|
|
||||||
|
// @block-start check-gate
|
||||||
|
async Task CheckGate()
|
||||||
|
{
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
|
||||||
|
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
|
||||||
|
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
|
||||||
|
|
||||||
|
if (model != MODEL || pipeline != PIPELINE)
|
||||||
|
throw new InvalidOperationException(
|
||||||
|
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
|
||||||
|
}
|
||||||
|
// @block-end check-gate
|
||||||
|
|
||||||
|
// @block-start identity-and-fingerprint
|
||||||
|
string ContentHash(string text) =>
|
||||||
|
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
|
||||||
|
|
||||||
|
// Qdrant accepts any well-formed UUID as a point ID:
|
||||||
|
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
|
||||||
|
string PointIdFor(string url, string anchor, int num) =>
|
||||||
|
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
|
||||||
|
{
|
||||||
|
var prepared = new List<Chunk>();
|
||||||
|
foreach (var c in chunks)
|
||||||
|
{
|
||||||
|
var text = Normalize(c.Text);
|
||||||
|
prepared.Add(c with
|
||||||
|
{
|
||||||
|
Text = text,
|
||||||
|
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
|
||||||
|
ContentHash = ContentHash(text),
|
||||||
|
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
return prepared;
|
||||||
|
}
|
||||||
|
// @block-end identity-and-fingerprint
|
||||||
|
|
||||||
|
// @block-start payload
|
||||||
|
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
|
||||||
|
{
|
||||||
|
["url"] = chunk.Url,
|
||||||
|
["anchor"] = chunk.Anchor,
|
||||||
|
["chunk_num"] = chunk.ChunkNum,
|
||||||
|
["section_url"] = chunk.SectionUrl,
|
||||||
|
["text"] = chunk.Text,
|
||||||
|
["content_hash"] = chunk.ContentHash,
|
||||||
|
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
|
||||||
|
};
|
||||||
|
// @block-end payload
|
||||||
|
|
||||||
|
// @block-start payload-indexes
|
||||||
|
foreach (var field in new[] { "content_hash", "url", "section_url" })
|
||||||
|
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
|
||||||
|
// @block-end payload-indexes
|
||||||
|
|
||||||
|
// @block-start populate
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
}).ToList(),
|
||||||
|
wait: true
|
||||||
|
);
|
||||||
|
// @block-end populate
|
||||||
|
|
||||||
|
// @block-start search
|
||||||
|
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
await client.QueryAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
query: new Document { Text = QUERY, Model = MODEL },
|
||||||
|
limit: 3,
|
||||||
|
payloadSelector: new[] { "section_url", "text" }
|
||||||
|
);
|
||||||
|
// @block-end search
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||||
|
var LATEST_CHUNKS = PrepareChunksForSync(CHUNKS);
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start split-by-state
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
|
||||||
|
SplitByState(List<Chunk> latestChunks)
|
||||||
|
{
|
||||||
|
var incoming = latestChunks.ToDictionary(c => c.PointId);
|
||||||
|
|
||||||
|
var stored = new Dictionary<string, string>();
|
||||||
|
var points = await client.RetrieveAsync(
|
||||||
|
COLLECTION,
|
||||||
|
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
|
||||||
|
payloadSelector: new[] { "content_hash" },
|
||||||
|
vectorSelector: false
|
||||||
|
);
|
||||||
|
foreach (var p in points)
|
||||||
|
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
|
||||||
|
|
||||||
|
var unchanged = new List<Chunk>();
|
||||||
|
var contentChanged = new List<Chunk>();
|
||||||
|
var unknownIds = new List<Chunk>();
|
||||||
|
foreach (var (pid, c) in incoming)
|
||||||
|
{
|
||||||
|
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
|
||||||
|
unchanged.Add(c);
|
||||||
|
else if (stored.ContainsKey(pid))
|
||||||
|
contentChanged.Add(c);
|
||||||
|
else
|
||||||
|
unknownIds.Add(c);
|
||||||
|
}
|
||||||
|
|
||||||
|
return (incoming, unchanged, contentChanged, unknownIds);
|
||||||
|
}
|
||||||
|
|
||||||
|
var splitState = await SplitByState(LATEST_CHUNKS);
|
||||||
|
// @block-end split-by-state
|
||||||
|
|
||||||
|
// @block-start re-embed-changed
|
||||||
|
async Task ReEmbedChanged(List<Chunk> contentChanged)
|
||||||
|
{
|
||||||
|
if (contentChanged.Count == 0)
|
||||||
|
return;
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
points: contentChanged.Select(c => new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
}).ToList(),
|
||||||
|
wait: true
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// @block-end re-embed-changed
|
||||||
|
|
||||||
|
// @block-start reuse-or-add
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
|
||||||
|
{
|
||||||
|
int reused = 0, added = 0;
|
||||||
|
|
||||||
|
foreach (var c in unknownIds)
|
||||||
|
{
|
||||||
|
var sameText = new Filter
|
||||||
|
{
|
||||||
|
Must = { MatchKeyword("content_hash", c.ContentHash) }
|
||||||
|
};
|
||||||
|
var hits = (await client.ScrollAsync(
|
||||||
|
COLLECTION,
|
||||||
|
filter: sameText,
|
||||||
|
limit: 1,
|
||||||
|
payloadSelector: new[] { "last_updated" },
|
||||||
|
vectorsSelector: true
|
||||||
|
)).Result;
|
||||||
|
|
||||||
|
PointStruct point;
|
||||||
|
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
|
||||||
|
{
|
||||||
|
point = new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
|
||||||
|
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
|
||||||
|
};
|
||||||
|
reused++;
|
||||||
|
}
|
||||||
|
else // genuinely new content: embed and insert
|
||||||
|
{
|
||||||
|
point = new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
};
|
||||||
|
added++;
|
||||||
|
}
|
||||||
|
|
||||||
|
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
|
||||||
|
}
|
||||||
|
|
||||||
|
return (reused, added);
|
||||||
|
}
|
||||||
|
// @block-end reuse-or-add
|
||||||
|
|
||||||
|
// @block-start delete-gone
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
|
||||||
|
{
|
||||||
|
if (incomingIds.Count == 0)
|
||||||
|
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
|
||||||
|
|
||||||
|
var stale = new Filter
|
||||||
|
{
|
||||||
|
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
|
||||||
|
};
|
||||||
|
|
||||||
|
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
// @block-end delete-gone
|
||||||
|
|
||||||
|
// @block-start sync
|
||||||
|
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
|
||||||
|
{
|
||||||
|
await CheckGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
var chunks = PrepareChunksForSync(latestChunks);
|
||||||
|
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
|
||||||
|
|
||||||
|
await ReEmbedChanged(contentChanged);
|
||||||
|
var (reused, added) = await ReuseOrAdd(unknownIds);
|
||||||
|
var deleted = await DeleteGone(incomingIds);
|
||||||
|
|
||||||
|
return new Dictionary<string, long>
|
||||||
|
{
|
||||||
|
["unchanged"] = unchanged.Count,
|
||||||
|
["re-embedded"] = contentChanged.Count,
|
||||||
|
["reused_embedding"] = reused,
|
||||||
|
["added"] = added,
|
||||||
|
["deleted"] = (long)deleted,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
// @block-end sync
|
||||||
|
|
||||||
|
// @block-start run-sync
|
||||||
|
var run = await Sync(LATEST_CHUNKS);
|
||||||
|
foreach (var (op, count) in run)
|
||||||
|
Console.WriteLine($"{op}: {count}");
|
||||||
|
// @block-end run-sync
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
+13
@@ -0,0 +1,13 @@
|
|||||||
|
```csharp
|
||||||
|
async Task CheckGate()
|
||||||
|
{
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
|
||||||
|
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
|
||||||
|
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
|
||||||
|
|
||||||
|
if (model != MODEL || pipeline != PIPELINE)
|
||||||
|
throw new InvalidOperationException(
|
||||||
|
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
|
||||||
|
}
|
||||||
|
```
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
```go
|
||||||
|
checkGate := func() {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
|
||||||
|
meta := info.GetConfig().GetMetadata()
|
||||||
|
|
||||||
|
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
|
||||||
|
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```java
|
||||||
|
static void checkGate() throws Exception {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
Map<String, Value> meta =
|
||||||
|
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
|
||||||
|
|
||||||
|
Value model = meta.get("embedding_model");
|
||||||
|
Value pipeline = meta.get("pipeline_version");
|
||||||
|
if (model == null || !MODEL.equals(model.getStringValue())
|
||||||
|
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
|
||||||
|
throw new RuntimeException(
|
||||||
|
"collection was built by " + meta + ": full re-embed into a fresh collection required");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+8
@@ -0,0 +1,8 @@
|
|||||||
|
```python
|
||||||
|
def check_gate():
|
||||||
|
# compare this pipeline's constants against what the collection records about itself
|
||||||
|
meta = client.get_collection(COLLECTION).config.metadata or {}
|
||||||
|
|
||||||
|
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
||||||
|
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
||||||
|
```
|
||||||
+22
@@ -0,0 +1,22 @@
|
|||||||
|
```rust
|
||||||
|
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
let meta = client
|
||||||
|
.collection_info(COLLECTION)
|
||||||
|
.await?
|
||||||
|
.result
|
||||||
|
.and_then(|info| info.config)
|
||||||
|
.map(|config| config.metadata)
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|
||||||
|
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
|
||||||
|
!= Some(PIPELINE)
|
||||||
|
{
|
||||||
|
anyhow::bail!(
|
||||||
|
"collection was built by {meta:?}: full re-embed into a fresh collection required"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
```
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
```typescript
|
||||||
|
async function checkGate() {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
|
||||||
|
{}) as Record<string, unknown>;
|
||||||
|
|
||||||
|
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
|
||||||
|
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```csharp
|
||||||
|
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
|
||||||
|
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
|
||||||
|
|
||||||
|
var client = new QdrantClient(
|
||||||
|
host: QDRANT_URL!,
|
||||||
|
https: true,
|
||||||
|
apiKey: QDRANT_API_KEY
|
||||||
|
);
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```go
|
||||||
|
QDRANT_URL := os.Getenv("QDRANT_URL")
|
||||||
|
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
|
||||||
|
|
||||||
|
client, err := qdrant.NewClient(&qdrant.Config{
|
||||||
|
Host: QDRANT_URL,
|
||||||
|
APIKey: QDRANT_API_KEY,
|
||||||
|
UseTLS: true,
|
||||||
|
})
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```java
|
||||||
|
static final String QDRANT_URL = System.getenv("QDRANT_URL");
|
||||||
|
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
|
||||||
|
|
||||||
|
static final QdrantClient client =
|
||||||
|
new QdrantClient(
|
||||||
|
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
|
||||||
|
.withApiKey(QDRANT_API_KEY)
|
||||||
|
.build());
|
||||||
|
```
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
```python
|
||||||
|
import os
|
||||||
|
|
||||||
|
from qdrant_client import QdrantClient, models
|
||||||
|
|
||||||
|
QDRANT_URL = os.getenv("QDRANT_URL")
|
||||||
|
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
||||||
|
|
||||||
|
client = QdrantClient(
|
||||||
|
url=QDRANT_URL,
|
||||||
|
api_key=QDRANT_API_KEY,
|
||||||
|
cloud_inference=True
|
||||||
|
)
|
||||||
|
```
|
||||||
+8
@@ -0,0 +1,8 @@
|
|||||||
|
```rust
|
||||||
|
let qdrant_url = std::env::var("QDRANT_URL")?;
|
||||||
|
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
|
||||||
|
|
||||||
|
let client = Qdrant::from_url(&qdrant_url)
|
||||||
|
.api_key(qdrant_api_key)
|
||||||
|
.build()?;
|
||||||
|
```
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
```typescript
|
||||||
|
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
|
||||||
|
|
||||||
|
const QDRANT_URL = process.env.QDRANT_URL;
|
||||||
|
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
|
||||||
|
|
||||||
|
const client = new QdrantClient({
|
||||||
|
url: QDRANT_URL,
|
||||||
|
apiKey: QDRANT_API_KEY,
|
||||||
|
});
|
||||||
|
```
|
||||||
+19
@@ -0,0 +1,19 @@
|
|||||||
|
```csharp
|
||||||
|
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
var PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
var COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
await client.CreateCollectionAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
vectorsConfig: new VectorParams
|
||||||
|
{
|
||||||
|
Size = 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance = Distance.Cosine
|
||||||
|
},
|
||||||
|
metadata: new()
|
||||||
|
{
|
||||||
|
["embedding_model"] = MODEL,
|
||||||
|
["pipeline_version"] = PIPELINE
|
||||||
|
}
|
||||||
|
);
|
||||||
|
```
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
```go
|
||||||
|
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
|
||||||
|
PIPELINE := "docs-prep-pipeline-v1"
|
||||||
|
COLLECTION := "docs-sync-tutorial"
|
||||||
|
|
||||||
|
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||||
|
Size: 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance: qdrant.Distance_Cosine,
|
||||||
|
}),
|
||||||
|
Metadata: qdrant.NewValueMap(map[string]any{
|
||||||
|
"embedding_model": MODEL,
|
||||||
|
"pipeline_version": PIPELINE,
|
||||||
|
}),
|
||||||
|
})
|
||||||
|
```
|
||||||
+24
@@ -0,0 +1,24 @@
|
|||||||
|
```java
|
||||||
|
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
static final String PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
static final String COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
static void createCollection() throws Exception {
|
||||||
|
client.createCollectionAsync(
|
||||||
|
CreateCollection.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setVectorsConfig(
|
||||||
|
VectorsConfig.newBuilder()
|
||||||
|
.setParams(
|
||||||
|
VectorParams.newBuilder()
|
||||||
|
.setSize(384) // all-MiniLM-L6-v2 output dimension
|
||||||
|
.setDistance(Distance.Cosine)
|
||||||
|
.build())
|
||||||
|
.build())
|
||||||
|
.putAllMetadata(
|
||||||
|
Map.of(
|
||||||
|
"embedding_model", value(MODEL),
|
||||||
|
"pipeline_version", value(PIPELINE)))
|
||||||
|
.build()).get();
|
||||||
|
}
|
||||||
|
```
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
```python
|
||||||
|
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
||||||
|
PIPELINE = "docs-prep-pipeline-v1"
|
||||||
|
COLLECTION = "docs-sync-tutorial"
|
||||||
|
|
||||||
|
client.create_collection(
|
||||||
|
COLLECTION,
|
||||||
|
vectors_config=models.VectorParams(
|
||||||
|
size=384, # all-MiniLM-L6-v2 output dimension
|
||||||
|
distance=models.Distance.COSINE,
|
||||||
|
),
|
||||||
|
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
||||||
|
)
|
||||||
|
```
|
||||||
+20
@@ -0,0 +1,20 @@
|
|||||||
|
```rust
|
||||||
|
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
const PIPELINE: &str = "docs-prep-pipeline-v1";
|
||||||
|
const COLLECTION: &str = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
let mut metadata: HashMap<String, Value> = HashMap::new();
|
||||||
|
metadata.insert("embedding_model".to_string(), json!(MODEL));
|
||||||
|
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
|
||||||
|
|
||||||
|
client
|
||||||
|
.create_collection(
|
||||||
|
CreateCollectionBuilder::new(COLLECTION)
|
||||||
|
.vectors_config(VectorParamsBuilder::new(
|
||||||
|
384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance::Cosine,
|
||||||
|
))
|
||||||
|
.metadata(metadata),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```typescript
|
||||||
|
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
const PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
const COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
await client.createCollection(COLLECTION, {
|
||||||
|
vectors: {
|
||||||
|
size: 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
distance: "Cosine",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
await client.updateCollection(COLLECTION, {
|
||||||
|
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
|
||||||
|
});
|
||||||
|
```
|
||||||
+248
@@ -0,0 +1,248 @@
|
|||||||
|
```csharp
|
||||||
|
using System.Security.Cryptography;
|
||||||
|
using System.Text;
|
||||||
|
using System.Text.RegularExpressions;
|
||||||
|
using Qdrant.Client;
|
||||||
|
using Qdrant.Client.Grpc;
|
||||||
|
using static Qdrant.Client.Grpc.Conditions;
|
||||||
|
using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
|
||||||
|
|
||||||
|
var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
|
||||||
|
var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
|
||||||
|
|
||||||
|
var client = new QdrantClient(
|
||||||
|
host: QDRANT_URL!,
|
||||||
|
https: true,
|
||||||
|
apiKey: QDRANT_API_KEY
|
||||||
|
);
|
||||||
|
|
||||||
|
var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
var PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
var COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
await client.CreateCollectionAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
vectorsConfig: new VectorParams
|
||||||
|
{
|
||||||
|
Size = 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance = Distance.Cosine
|
||||||
|
},
|
||||||
|
metadata: new()
|
||||||
|
{
|
||||||
|
["embedding_model"] = MODEL,
|
||||||
|
["pipeline_version"] = PIPELINE
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
async Task CheckGate()
|
||||||
|
{
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
|
||||||
|
var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
|
||||||
|
var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
|
||||||
|
|
||||||
|
if (model != MODEL || pipeline != PIPELINE)
|
||||||
|
throw new InvalidOperationException(
|
||||||
|
$"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
|
||||||
|
}
|
||||||
|
|
||||||
|
string ContentHash(string text) =>
|
||||||
|
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
|
||||||
|
|
||||||
|
// Qdrant accepts any well-formed UUID as a point ID:
|
||||||
|
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
|
||||||
|
string PointIdFor(string url, string anchor, int num) =>
|
||||||
|
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
|
||||||
|
{
|
||||||
|
var prepared = new List<Chunk>();
|
||||||
|
foreach (var c in chunks)
|
||||||
|
{
|
||||||
|
var text = Normalize(c.Text);
|
||||||
|
prepared.Add(c with
|
||||||
|
{
|
||||||
|
Text = text,
|
||||||
|
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
|
||||||
|
ContentHash = ContentHash(text),
|
||||||
|
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
return prepared;
|
||||||
|
}
|
||||||
|
|
||||||
|
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
|
||||||
|
{
|
||||||
|
["url"] = chunk.Url,
|
||||||
|
["anchor"] = chunk.Anchor,
|
||||||
|
["chunk_num"] = chunk.ChunkNum,
|
||||||
|
["section_url"] = chunk.SectionUrl,
|
||||||
|
["text"] = chunk.Text,
|
||||||
|
["content_hash"] = chunk.ContentHash,
|
||||||
|
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
|
||||||
|
};
|
||||||
|
|
||||||
|
foreach (var field in new[] { "content_hash", "url", "section_url" })
|
||||||
|
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
|
||||||
|
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
}).ToList(),
|
||||||
|
wait: true
|
||||||
|
);
|
||||||
|
|
||||||
|
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
await client.QueryAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
query: new Document { Text = QUERY, Model = MODEL },
|
||||||
|
limit: 3,
|
||||||
|
payloadSelector: new[] { "section_url", "text" }
|
||||||
|
);
|
||||||
|
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
|
||||||
|
SplitByState(List<Chunk> latestChunks)
|
||||||
|
{
|
||||||
|
var incoming = latestChunks.ToDictionary(c => c.PointId);
|
||||||
|
|
||||||
|
var stored = new Dictionary<string, string>();
|
||||||
|
var points = await client.RetrieveAsync(
|
||||||
|
COLLECTION,
|
||||||
|
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
|
||||||
|
payloadSelector: new[] { "content_hash" },
|
||||||
|
vectorSelector: false
|
||||||
|
);
|
||||||
|
foreach (var p in points)
|
||||||
|
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
|
||||||
|
|
||||||
|
var unchanged = new List<Chunk>();
|
||||||
|
var contentChanged = new List<Chunk>();
|
||||||
|
var unknownIds = new List<Chunk>();
|
||||||
|
foreach (var (pid, c) in incoming)
|
||||||
|
{
|
||||||
|
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
|
||||||
|
unchanged.Add(c);
|
||||||
|
else if (stored.ContainsKey(pid))
|
||||||
|
contentChanged.Add(c);
|
||||||
|
else
|
||||||
|
unknownIds.Add(c);
|
||||||
|
}
|
||||||
|
|
||||||
|
return (incoming, unchanged, contentChanged, unknownIds);
|
||||||
|
}
|
||||||
|
|
||||||
|
var splitState = await SplitByState(LATEST_CHUNKS);
|
||||||
|
|
||||||
|
async Task ReEmbedChanged(List<Chunk> contentChanged)
|
||||||
|
{
|
||||||
|
if (contentChanged.Count == 0)
|
||||||
|
return;
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
points: contentChanged.Select(c => new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
}).ToList(),
|
||||||
|
wait: true
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
|
||||||
|
{
|
||||||
|
int reused = 0, added = 0;
|
||||||
|
|
||||||
|
foreach (var c in unknownIds)
|
||||||
|
{
|
||||||
|
var sameText = new Filter
|
||||||
|
{
|
||||||
|
Must = { MatchKeyword("content_hash", c.ContentHash) }
|
||||||
|
};
|
||||||
|
var hits = (await client.ScrollAsync(
|
||||||
|
COLLECTION,
|
||||||
|
filter: sameText,
|
||||||
|
limit: 1,
|
||||||
|
payloadSelector: new[] { "last_updated" },
|
||||||
|
vectorsSelector: true
|
||||||
|
)).Result;
|
||||||
|
|
||||||
|
PointStruct point;
|
||||||
|
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
|
||||||
|
{
|
||||||
|
point = new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
|
||||||
|
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
|
||||||
|
};
|
||||||
|
reused++;
|
||||||
|
}
|
||||||
|
else // genuinely new content: embed and insert
|
||||||
|
{
|
||||||
|
point = new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
};
|
||||||
|
added++;
|
||||||
|
}
|
||||||
|
|
||||||
|
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
|
||||||
|
}
|
||||||
|
|
||||||
|
return (reused, added);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
|
||||||
|
{
|
||||||
|
if (incomingIds.Count == 0)
|
||||||
|
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
|
||||||
|
|
||||||
|
var stale = new Filter
|
||||||
|
{
|
||||||
|
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
|
||||||
|
};
|
||||||
|
|
||||||
|
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
|
||||||
|
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
|
||||||
|
{
|
||||||
|
await CheckGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
var chunks = PrepareChunksForSync(latestChunks);
|
||||||
|
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
|
||||||
|
|
||||||
|
await ReEmbedChanged(contentChanged);
|
||||||
|
var (reused, added) = await ReuseOrAdd(unknownIds);
|
||||||
|
var deleted = await DeleteGone(incomingIds);
|
||||||
|
|
||||||
|
return new Dictionary<string, long>
|
||||||
|
{
|
||||||
|
["unchanged"] = unchanged.Count,
|
||||||
|
["re-embedded"] = contentChanged.Count,
|
||||||
|
["reused_embedding"] = reused,
|
||||||
|
["added"] = added,
|
||||||
|
["deleted"] = (long)deleted,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
var run = await Sync(LATEST_CHUNKS);
|
||||||
|
foreach (var (op, count) in run)
|
||||||
|
Console.WriteLine($"{op}: {count}");
|
||||||
|
```
|
||||||
+19
@@ -0,0 +1,19 @@
|
|||||||
|
```csharp
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async Task<ulong> DeleteGone(Dictionary<string, Chunk> incomingIds)
|
||||||
|
{
|
||||||
|
if (incomingIds.Count == 0)
|
||||||
|
throw new ArgumentException("Refusing to delete from an empty source snapshot.");
|
||||||
|
|
||||||
|
var stale = new Filter
|
||||||
|
{
|
||||||
|
MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
|
||||||
|
};
|
||||||
|
|
||||||
|
var toDelete = await client.CountAsync(COLLECTION, filter: stale);
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
```
|
||||||
+29
@@ -0,0 +1,29 @@
|
|||||||
|
```go
|
||||||
|
// remove every point the current crawl no longer contains, return how many
|
||||||
|
deleteGone := func(incomingIDs map[string]Chunk) int {
|
||||||
|
if len(incomingIDs) == 0 {
|
||||||
|
panic("Refusing to delete from an empty source snapshot.")
|
||||||
|
}
|
||||||
|
|
||||||
|
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
|
||||||
|
for pid := range incomingIDs {
|
||||||
|
ids = append(ids, qdrant.NewID(pid))
|
||||||
|
}
|
||||||
|
stale := &qdrant.Filter{
|
||||||
|
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
|
||||||
|
}
|
||||||
|
|
||||||
|
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Filter: stale,
|
||||||
|
})
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.Delete(context.Background(), &qdrant.DeletePoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: qdrant.NewPointsSelectorFilter(stale),
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
return int(toDelete)
|
||||||
|
}
|
||||||
|
```
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
```java
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
|
||||||
|
if (incomingIds.isEmpty()) {
|
||||||
|
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
Filter stale = Filter.newBuilder()
|
||||||
|
.addMustNot(hasId(
|
||||||
|
incomingIds.keySet().stream()
|
||||||
|
.map(pid -> id(UUID.fromString(pid)))
|
||||||
|
.collect(Collectors.toList())))
|
||||||
|
.build();
|
||||||
|
|
||||||
|
long toDelete = client.countAsync(COLLECTION, stale, true).get();
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.deleteAsync(COLLECTION, stale).get();
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
```
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
```python
|
||||||
|
def delete_gone(incoming_ids):
|
||||||
|
"""Remove every point the current crawl no longer contains. Returns how many."""
|
||||||
|
if not incoming_ids:
|
||||||
|
raise ValueError("Refusing to delete from an empty source snapshot.")
|
||||||
|
|
||||||
|
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
||||||
|
|
||||||
|
to_delete = client.count(COLLECTION, count_filter=stale).count
|
||||||
|
|
||||||
|
# potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
||||||
|
return to_delete
|
||||||
|
```
|
||||||
+28
@@ -0,0 +1,28 @@
|
|||||||
|
```rust
|
||||||
|
/// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async fn delete_gone(
|
||||||
|
client: &Qdrant,
|
||||||
|
incoming_ids: &HashMap<String, Chunk>,
|
||||||
|
) -> anyhow::Result<u64> {
|
||||||
|
if incoming_ids.is_empty() {
|
||||||
|
anyhow::bail!("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
let stale = Filter::must_not([Condition::has_id(
|
||||||
|
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
|
||||||
|
)]);
|
||||||
|
|
||||||
|
let to_delete = client
|
||||||
|
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
|
||||||
|
.await?
|
||||||
|
.result
|
||||||
|
.map(|r| r.count)
|
||||||
|
.unwrap_or(0);
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client
|
||||||
|
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
|
||||||
|
.await?;
|
||||||
|
Ok(to_delete)
|
||||||
|
}
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```typescript
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async function deleteGone(incoming: Map<string, SyncChunk>) {
|
||||||
|
if (incoming.size === 0) {
|
||||||
|
throw new Error("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
|
||||||
|
|
||||||
|
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
await client.delete(COLLECTION, { filter: stale, wait: true });
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
```
|
||||||
+276
@@ -0,0 +1,276 @@
|
|||||||
|
```go
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"crypto/sha256"
|
||||||
|
"encoding/hex"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/google/uuid"
|
||||||
|
"github.com/qdrant/go-client/qdrant"
|
||||||
|
)
|
||||||
|
|
||||||
|
QDRANT_URL := os.Getenv("QDRANT_URL")
|
||||||
|
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
|
||||||
|
|
||||||
|
client, err := qdrant.NewClient(&qdrant.Config{
|
||||||
|
Host: QDRANT_URL,
|
||||||
|
APIKey: QDRANT_API_KEY,
|
||||||
|
UseTLS: true,
|
||||||
|
})
|
||||||
|
|
||||||
|
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
|
||||||
|
PIPELINE := "docs-prep-pipeline-v1"
|
||||||
|
COLLECTION := "docs-sync-tutorial"
|
||||||
|
|
||||||
|
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||||
|
Size: 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance: qdrant.Distance_Cosine,
|
||||||
|
}),
|
||||||
|
Metadata: qdrant.NewValueMap(map[string]any{
|
||||||
|
"embedding_model": MODEL,
|
||||||
|
"pipeline_version": PIPELINE,
|
||||||
|
}),
|
||||||
|
})
|
||||||
|
|
||||||
|
checkGate := func() {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
|
||||||
|
meta := info.GetConfig().GetMetadata()
|
||||||
|
|
||||||
|
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
|
||||||
|
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
contentHash := func(text string) string {
|
||||||
|
sum := sha256.Sum256([]byte(text))
|
||||||
|
return hex.EncodeToString(sum[:])
|
||||||
|
}
|
||||||
|
|
||||||
|
pointID := func(url, anchor string, num int) string {
|
||||||
|
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
|
||||||
|
// marking the input as a URL-like name
|
||||||
|
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// derive both values (and the section address) for every raw chunk
|
||||||
|
prepareChunksForSync := func(chunks []Chunk) []Chunk {
|
||||||
|
out := make([]Chunk, 0, len(chunks))
|
||||||
|
for _, c := range chunks {
|
||||||
|
c.Text = normalize(c.Text)
|
||||||
|
c.SectionURL = c.URL
|
||||||
|
if c.Anchor != "" {
|
||||||
|
c.SectionURL = c.URL + "#" + c.Anchor
|
||||||
|
}
|
||||||
|
c.ContentHash = contentHash(c.Text)
|
||||||
|
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
|
||||||
|
out = append(out, c)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
payload := func(c Chunk, lastUpdated string) map[string]any {
|
||||||
|
if lastUpdated == "" {
|
||||||
|
lastUpdated = time.Now().UTC().Format(time.RFC3339)
|
||||||
|
}
|
||||||
|
return map[string]any{
|
||||||
|
"url": c.URL,
|
||||||
|
"anchor": c.Anchor,
|
||||||
|
"chunk_num": c.ChunkNum,
|
||||||
|
"section_url": c.SectionURL,
|
||||||
|
"text": c.Text,
|
||||||
|
"content_hash": c.ContentHash,
|
||||||
|
"last_updated": lastUpdated,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, field := range []string{"content_hash", "url", "section_url"} {
|
||||||
|
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
FieldName: field,
|
||||||
|
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
var points []*qdrant.PointStruct
|
||||||
|
for _, c := range prepareChunksForSync(CHUNKS) {
|
||||||
|
points = append(points, &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
|
||||||
|
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||||
|
|
||||||
|
client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(3)),
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
|
||||||
|
})
|
||||||
|
|
||||||
|
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
|
||||||
|
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
|
||||||
|
incoming := make(map[string]Chunk, len(latestChunks))
|
||||||
|
ids := make([]*qdrant.PointId, 0, len(latestChunks))
|
||||||
|
for _, c := range latestChunks {
|
||||||
|
incoming[c.PointID] = c
|
||||||
|
ids = append(ids, qdrant.NewID(c.PointID))
|
||||||
|
}
|
||||||
|
|
||||||
|
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Ids: ids,
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
|
||||||
|
WithVectors: qdrant.NewWithVectors(false),
|
||||||
|
})
|
||||||
|
stored := make(map[string]string, len(retrieved))
|
||||||
|
for _, p := range retrieved {
|
||||||
|
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
|
||||||
|
}
|
||||||
|
|
||||||
|
var unchanged, contentChanged, unknownIDs []Chunk
|
||||||
|
for pid, c := range incoming {
|
||||||
|
storedHash, found := stored[pid]
|
||||||
|
switch {
|
||||||
|
case found && storedHash == c.ContentHash:
|
||||||
|
unchanged = append(unchanged, c)
|
||||||
|
case found:
|
||||||
|
contentChanged = append(contentChanged, c)
|
||||||
|
default:
|
||||||
|
unknownIDs = append(unknownIDs, c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return incoming, unchanged, contentChanged, unknownIDs
|
||||||
|
}
|
||||||
|
|
||||||
|
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
|
||||||
|
|
||||||
|
reEmbedChanged := func(contentChanged []Chunk) {
|
||||||
|
if len(contentChanged) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
|
||||||
|
for _, c := range contentChanged {
|
||||||
|
points = append(points, &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// reuse an existing embedding when the same text is already stored; embed only what is new
|
||||||
|
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
|
||||||
|
reused, added := 0, 0
|
||||||
|
|
||||||
|
for _, c := range unknownIDs {
|
||||||
|
sameText := &qdrant.Filter{
|
||||||
|
Must: []*qdrant.Condition{
|
||||||
|
qdrant.NewMatch("content_hash", c.ContentHash),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Filter: sameText,
|
||||||
|
Limit: qdrant.PtrOf(uint32(1)),
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
|
||||||
|
WithVectors: qdrant.NewWithVectors(true),
|
||||||
|
})
|
||||||
|
|
||||||
|
var point *qdrant.PointStruct
|
||||||
|
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
|
||||||
|
}
|
||||||
|
reused++
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
}
|
||||||
|
added++
|
||||||
|
}
|
||||||
|
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{point},
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
return reused, added
|
||||||
|
}
|
||||||
|
|
||||||
|
// remove every point the current crawl no longer contains, return how many
|
||||||
|
deleteGone := func(incomingIDs map[string]Chunk) int {
|
||||||
|
if len(incomingIDs) == 0 {
|
||||||
|
panic("Refusing to delete from an empty source snapshot.")
|
||||||
|
}
|
||||||
|
|
||||||
|
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
|
||||||
|
for pid := range incomingIDs {
|
||||||
|
ids = append(ids, qdrant.NewID(pid))
|
||||||
|
}
|
||||||
|
stale := &qdrant.Filter{
|
||||||
|
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
|
||||||
|
}
|
||||||
|
|
||||||
|
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Filter: stale,
|
||||||
|
})
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.Delete(context.Background(), &qdrant.DeletePoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: qdrant.NewPointsSelectorFilter(stale),
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
return int(toDelete)
|
||||||
|
}
|
||||||
|
|
||||||
|
sync := func(latestChunks []Chunk) map[string]int {
|
||||||
|
checkGate() // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
chunks := prepareChunksForSync(latestChunks)
|
||||||
|
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
|
||||||
|
|
||||||
|
reEmbedChanged(contentChanged)
|
||||||
|
reused, added := reuseOrAdd(unknownIDs)
|
||||||
|
deleted := deleteGone(incomingIDs)
|
||||||
|
|
||||||
|
return map[string]int{
|
||||||
|
"unchanged": len(unchanged),
|
||||||
|
"re-embedded": len(contentChanged),
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
run := sync(LATEST_CHUNKS)
|
||||||
|
fmt.Println(run)
|
||||||
|
```
|
||||||
+27
@@ -0,0 +1,27 @@
|
|||||||
|
```csharp
|
||||||
|
string ContentHash(string text) =>
|
||||||
|
Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
|
||||||
|
|
||||||
|
// Qdrant accepts any well-formed UUID as a point ID:
|
||||||
|
// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
|
||||||
|
string PointIdFor(string url, string anchor, int num) =>
|
||||||
|
new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
List<Chunk> PrepareChunksForSync(List<Chunk> chunks)
|
||||||
|
{
|
||||||
|
var prepared = new List<Chunk>();
|
||||||
|
foreach (var c in chunks)
|
||||||
|
{
|
||||||
|
var text = Normalize(c.Text);
|
||||||
|
prepared.Add(c with
|
||||||
|
{
|
||||||
|
Text = text,
|
||||||
|
SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
|
||||||
|
ContentHash = ContentHash(text),
|
||||||
|
PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
return prepared;
|
||||||
|
}
|
||||||
|
```
|
||||||
+28
@@ -0,0 +1,28 @@
|
|||||||
|
```go
|
||||||
|
contentHash := func(text string) string {
|
||||||
|
sum := sha256.Sum256([]byte(text))
|
||||||
|
return hex.EncodeToString(sum[:])
|
||||||
|
}
|
||||||
|
|
||||||
|
pointID := func(url, anchor string, num int) string {
|
||||||
|
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
|
||||||
|
// marking the input as a URL-like name
|
||||||
|
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// derive both values (and the section address) for every raw chunk
|
||||||
|
prepareChunksForSync := func(chunks []Chunk) []Chunk {
|
||||||
|
out := make([]Chunk, 0, len(chunks))
|
||||||
|
for _, c := range chunks {
|
||||||
|
c.Text = normalize(c.Text)
|
||||||
|
c.SectionURL = c.URL
|
||||||
|
if c.Anchor != "" {
|
||||||
|
c.SectionURL = c.URL + "#" + c.Anchor
|
||||||
|
}
|
||||||
|
c.ContentHash = contentHash(c.Text)
|
||||||
|
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
|
||||||
|
out = append(out, c)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
```
|
||||||
+27
@@ -0,0 +1,27 @@
|
|||||||
|
```java
|
||||||
|
static String contentHash(String text) throws Exception {
|
||||||
|
byte[] digest = MessageDigest.getInstance("SHA-256")
|
||||||
|
.digest(text.getBytes(StandardCharsets.UTF_8));
|
||||||
|
return String.format("%064x", new BigInteger(1, digest));
|
||||||
|
}
|
||||||
|
|
||||||
|
static String pointId(String url, String anchor, int num) {
|
||||||
|
// name-based UUID (version 3); the same address always yields the same ID
|
||||||
|
return UUID.nameUUIDFromBytes(
|
||||||
|
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
|
||||||
|
List<Chunk> out = new ArrayList<>();
|
||||||
|
for (Chunk c : chunks) {
|
||||||
|
String text = normalize(c.text);
|
||||||
|
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
|
||||||
|
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
|
||||||
|
prepared.contentHash = contentHash(text);
|
||||||
|
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
|
||||||
|
out.add(prepared);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
```
|
||||||
+26
@@ -0,0 +1,26 @@
|
|||||||
|
```python
|
||||||
|
import hashlib
|
||||||
|
import uuid
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
def content_hash(text):
|
||||||
|
return hashlib.sha256(text.encode()).hexdigest()
|
||||||
|
|
||||||
|
def point_id(url, anchor, num):
|
||||||
|
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||||
|
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
||||||
|
|
||||||
|
def prepare_chunks_for_sync(chunks):
|
||||||
|
"""Derive both values (and the section address) for every raw chunk."""
|
||||||
|
out = []
|
||||||
|
for c in chunks:
|
||||||
|
text = normalize(c["text"])
|
||||||
|
out.append({
|
||||||
|
**c,
|
||||||
|
"text": text,
|
||||||
|
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
||||||
|
"content_hash": content_hash(text),
|
||||||
|
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
```
|
||||||
+38
@@ -0,0 +1,38 @@
|
|||||||
|
```rust
|
||||||
|
fn content_hash(text: &str) -> String {
|
||||||
|
Sha256::digest(text.as_bytes())
|
||||||
|
.iter()
|
||||||
|
.map(|byte| format!("{byte:02x}"))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn point_id(url: &str, anchor: &str, num: u32) -> String {
|
||||||
|
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||||
|
uuid::Uuid::new_v5(
|
||||||
|
&uuid::Uuid::NAMESPACE_URL,
|
||||||
|
format!("{url}#{anchor}::{num}").as_bytes(),
|
||||||
|
)
|
||||||
|
.to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Derive both values (and the section address) for every raw chunk.
|
||||||
|
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
|
||||||
|
chunks
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
let text = normalize(&c.text);
|
||||||
|
Chunk {
|
||||||
|
text: text.clone(),
|
||||||
|
section_url: if c.anchor.is_empty() {
|
||||||
|
c.url.clone()
|
||||||
|
} else {
|
||||||
|
format!("{}#{}", c.url, c.anchor)
|
||||||
|
},
|
||||||
|
content_hash: content_hash(&text),
|
||||||
|
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
|
||||||
|
..c.clone()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
```
|
||||||
+32
@@ -0,0 +1,32 @@
|
|||||||
|
```typescript
|
||||||
|
import { createHash } from "node:crypto";
|
||||||
|
|
||||||
|
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
|
||||||
|
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
|
||||||
|
|
||||||
|
function contentHash(text: string): string {
|
||||||
|
return createHash("sha256").update(text).digest("hex");
|
||||||
|
}
|
||||||
|
|
||||||
|
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
|
||||||
|
function pointId(url: string, anchor: string, num: number): string {
|
||||||
|
// Qdrant accepts any well-formed UUID as a point ID:
|
||||||
|
// hash the address, format the digest as a UUID, and the same address always yields the same ID
|
||||||
|
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
|
||||||
|
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
|
||||||
|
return chunks.map((c) => {
|
||||||
|
const text = normalize(c.text);
|
||||||
|
return {
|
||||||
|
...c,
|
||||||
|
text,
|
||||||
|
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
|
||||||
|
content_hash: contentHash(text),
|
||||||
|
point_id: pointId(c.url, c.anchor, c.chunk_num),
|
||||||
|
};
|
||||||
|
});
|
||||||
|
}
|
||||||
|
```
|
||||||
+328
@@ -0,0 +1,328 @@
|
|||||||
|
```java
|
||||||
|
import static io.qdrant.client.ConditionFactory.hasId;
|
||||||
|
import static io.qdrant.client.ConditionFactory.matchKeyword;
|
||||||
|
import static io.qdrant.client.PointIdFactory.id;
|
||||||
|
import static io.qdrant.client.QueryFactory.nearest;
|
||||||
|
import static io.qdrant.client.ValueFactory.value;
|
||||||
|
import static io.qdrant.client.VectorFactory.vector;
|
||||||
|
import static io.qdrant.client.VectorsFactory.vectors;
|
||||||
|
|
||||||
|
import io.qdrant.client.QdrantClient;
|
||||||
|
import io.qdrant.client.QdrantGrpcClient;
|
||||||
|
import io.qdrant.client.VectorOutputHelper;
|
||||||
|
import io.qdrant.client.WithPayloadSelectorFactory;
|
||||||
|
import io.qdrant.client.WithVectorsSelectorFactory;
|
||||||
|
import io.qdrant.client.grpc.Collections.CreateCollection;
|
||||||
|
import io.qdrant.client.grpc.Collections.Distance;
|
||||||
|
import io.qdrant.client.grpc.Collections.PayloadSchemaType;
|
||||||
|
import io.qdrant.client.grpc.Collections.VectorParams;
|
||||||
|
import io.qdrant.client.grpc.Collections.VectorsConfig;
|
||||||
|
import io.qdrant.client.grpc.Common.Filter;
|
||||||
|
import io.qdrant.client.grpc.JsonWithInt.Value;
|
||||||
|
import io.qdrant.client.grpc.Points.Document;
|
||||||
|
import io.qdrant.client.grpc.Points.PointStruct;
|
||||||
|
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.ScrollPoints;
|
||||||
|
import java.math.BigInteger;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.security.MessageDigest;
|
||||||
|
import java.time.OffsetDateTime;
|
||||||
|
import java.time.ZoneOffset;
|
||||||
|
import java.time.temporal.ChronoUnit;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.UUID;
|
||||||
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
|
static final String QDRANT_URL = System.getenv("QDRANT_URL");
|
||||||
|
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
|
||||||
|
|
||||||
|
static final QdrantClient client =
|
||||||
|
new QdrantClient(
|
||||||
|
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
|
||||||
|
.withApiKey(QDRANT_API_KEY)
|
||||||
|
.build());
|
||||||
|
|
||||||
|
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
static final String PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
static final String COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
static void createCollection() throws Exception {
|
||||||
|
client.createCollectionAsync(
|
||||||
|
CreateCollection.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setVectorsConfig(
|
||||||
|
VectorsConfig.newBuilder()
|
||||||
|
.setParams(
|
||||||
|
VectorParams.newBuilder()
|
||||||
|
.setSize(384) // all-MiniLM-L6-v2 output dimension
|
||||||
|
.setDistance(Distance.Cosine)
|
||||||
|
.build())
|
||||||
|
.build())
|
||||||
|
.putAllMetadata(
|
||||||
|
Map.of(
|
||||||
|
"embedding_model", value(MODEL),
|
||||||
|
"pipeline_version", value(PIPELINE)))
|
||||||
|
.build()).get();
|
||||||
|
}
|
||||||
|
|
||||||
|
static void checkGate() throws Exception {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
Map<String, Value> meta =
|
||||||
|
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
|
||||||
|
|
||||||
|
Value model = meta.get("embedding_model");
|
||||||
|
Value pipeline = meta.get("pipeline_version");
|
||||||
|
if (model == null || !MODEL.equals(model.getStringValue())
|
||||||
|
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
|
||||||
|
throw new RuntimeException(
|
||||||
|
"collection was built by " + meta + ": full re-embed into a fresh collection required");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static String contentHash(String text) throws Exception {
|
||||||
|
byte[] digest = MessageDigest.getInstance("SHA-256")
|
||||||
|
.digest(text.getBytes(StandardCharsets.UTF_8));
|
||||||
|
return String.format("%064x", new BigInteger(1, digest));
|
||||||
|
}
|
||||||
|
|
||||||
|
static String pointId(String url, String anchor, int num) {
|
||||||
|
// name-based UUID (version 3); the same address always yields the same ID
|
||||||
|
return UUID.nameUUIDFromBytes(
|
||||||
|
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
|
||||||
|
List<Chunk> out = new ArrayList<>();
|
||||||
|
for (Chunk c : chunks) {
|
||||||
|
String text = normalize(c.text);
|
||||||
|
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
|
||||||
|
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
|
||||||
|
prepared.contentHash = contentHash(text);
|
||||||
|
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
|
||||||
|
out.add(prepared);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
|
||||||
|
Map<String, Value> p = new HashMap<>();
|
||||||
|
p.put("url", value(chunk.url));
|
||||||
|
p.put("anchor", value(chunk.anchor));
|
||||||
|
p.put("chunk_num", value(chunk.chunkNum));
|
||||||
|
p.put("section_url", value(chunk.sectionUrl));
|
||||||
|
p.put("text", value(chunk.text));
|
||||||
|
p.put("content_hash", value(chunk.contentHash));
|
||||||
|
p.put("last_updated", value(lastUpdated != null
|
||||||
|
? lastUpdated
|
||||||
|
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
|
||||||
|
return p;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void createPayloadIndexes() throws Exception {
|
||||||
|
for (String field : List.of("content_hash", "url", "section_url")) {
|
||||||
|
client.createPayloadIndexAsync(
|
||||||
|
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static void populate() throws Exception {
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (Chunk c : prepareChunksForSync(CHUNKS)) {
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
client.upsertAsync(COLLECTION, points).get();
|
||||||
|
}
|
||||||
|
|
||||||
|
static final String QUERY =
|
||||||
|
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
static void search() throws Exception {
|
||||||
|
client.queryAsync(
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(QUERY)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(3)
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
|
||||||
|
.build()).get();
|
||||||
|
}
|
||||||
|
|
||||||
|
static class SyncState {
|
||||||
|
Map<String, Chunk> incoming = new LinkedHashMap<>();
|
||||||
|
List<Chunk> unchanged = new ArrayList<>();
|
||||||
|
List<Chunk> contentChanged = new ArrayList<>();
|
||||||
|
List<Chunk> unknownIds = new ArrayList<>();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
|
||||||
|
SyncState state = new SyncState();
|
||||||
|
for (Chunk c : latestChunks) {
|
||||||
|
state.incoming.put(c.pointId, c);
|
||||||
|
}
|
||||||
|
|
||||||
|
Map<String, String> stored = new HashMap<>();
|
||||||
|
var points = client.retrieveAsync(
|
||||||
|
COLLECTION,
|
||||||
|
state.incoming.keySet().stream()
|
||||||
|
.map(pid -> id(UUID.fromString(pid)))
|
||||||
|
.collect(Collectors.toList()),
|
||||||
|
WithPayloadSelectorFactory.include(List.of("content_hash")),
|
||||||
|
WithVectorsSelectorFactory.enable(false),
|
||||||
|
null).get();
|
||||||
|
for (var p : points) {
|
||||||
|
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
|
||||||
|
}
|
||||||
|
|
||||||
|
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
|
||||||
|
String pid = e.getKey();
|
||||||
|
Chunk c = e.getValue();
|
||||||
|
if (c.contentHash.equals(stored.get(pid))) {
|
||||||
|
state.unchanged.add(c);
|
||||||
|
} else if (stored.containsKey(pid)) {
|
||||||
|
state.contentChanged.add(c);
|
||||||
|
} else {
|
||||||
|
state.unknownIds.add(c);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return state;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
|
||||||
|
if (contentChanged.isEmpty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (Chunk c : contentChanged) {
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
client.upsertAsync(COLLECTION, points).get();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
|
||||||
|
int reused = 0;
|
||||||
|
int added = 0;
|
||||||
|
|
||||||
|
for (Chunk c : unknownIds) {
|
||||||
|
Filter sameText = Filter.newBuilder()
|
||||||
|
.addMust(matchKeyword("content_hash", c.contentHash))
|
||||||
|
.build();
|
||||||
|
|
||||||
|
var hits = client.scrollAsync(
|
||||||
|
ScrollPoints.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setFilter(sameText)
|
||||||
|
.setLimit(1)
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
|
||||||
|
.setWithVectors(WithVectorsSelectorFactory.enable(true))
|
||||||
|
.build()).get().getResultList();
|
||||||
|
|
||||||
|
PointStruct point;
|
||||||
|
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(vectors(vector(
|
||||||
|
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
|
||||||
|
.getDataList())))
|
||||||
|
.putAllPayload(
|
||||||
|
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
|
||||||
|
.build();
|
||||||
|
reused++;
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build();
|
||||||
|
added++;
|
||||||
|
}
|
||||||
|
|
||||||
|
client.upsertAsync(COLLECTION, List.of(point)).get();
|
||||||
|
}
|
||||||
|
|
||||||
|
return new int[] {reused, added};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
|
||||||
|
if (incomingIds.isEmpty()) {
|
||||||
|
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
Filter stale = Filter.newBuilder()
|
||||||
|
.addMustNot(hasId(
|
||||||
|
incomingIds.keySet().stream()
|
||||||
|
.map(pid -> id(UUID.fromString(pid)))
|
||||||
|
.collect(Collectors.toList())))
|
||||||
|
.build();
|
||||||
|
|
||||||
|
long toDelete = client.countAsync(COLLECTION, stale, true).get();
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.deleteAsync(COLLECTION, stale).get();
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
|
||||||
|
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
|
||||||
|
checkGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
List<Chunk> chunks = prepareChunksForSync(latestChunks);
|
||||||
|
SyncState state = splitByState(chunks);
|
||||||
|
|
||||||
|
reEmbedChanged(state.contentChanged);
|
||||||
|
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
|
||||||
|
long deleted = deleteGone(state.incoming);
|
||||||
|
|
||||||
|
return Map.of(
|
||||||
|
"unchanged", (long) state.unchanged.size(),
|
||||||
|
"re-embedded", (long) state.contentChanged.size(),
|
||||||
|
"reused_embedding", (long) reusedAdded[0],
|
||||||
|
"added", (long) reusedAdded[1],
|
||||||
|
"deleted", deleted);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void runSync() throws Exception {
|
||||||
|
Map<String, Long> run = sync(LATEST_CHUNKS);
|
||||||
|
System.out.println(run);
|
||||||
|
}
|
||||||
|
```
|
||||||
+4
@@ -0,0 +1,4 @@
|
|||||||
|
```csharp
|
||||||
|
foreach (var field in new[] { "content_hash", "url", "section_url" })
|
||||||
|
await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
|
||||||
|
```
|
||||||
+9
@@ -0,0 +1,9 @@
|
|||||||
|
```go
|
||||||
|
for _, field := range []string{"content_hash", "url", "section_url"} {
|
||||||
|
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
FieldName: field,
|
||||||
|
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
```
|
||||||
+8
@@ -0,0 +1,8 @@
|
|||||||
|
```java
|
||||||
|
static void createPayloadIndexes() throws Exception {
|
||||||
|
for (String field : List.of("content_hash", "url", "section_url")) {
|
||||||
|
client.createPayloadIndexAsync(
|
||||||
|
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+4
@@ -0,0 +1,4 @@
|
|||||||
|
```python
|
||||||
|
for field in ("content_hash", "url", "section_url"):
|
||||||
|
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
||||||
|
```
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
```rust
|
||||||
|
for field in ["content_hash", "url", "section_url"] {
|
||||||
|
client
|
||||||
|
.create_field_index(CreateFieldIndexCollectionBuilder::new(
|
||||||
|
COLLECTION,
|
||||||
|
field,
|
||||||
|
FieldType::Keyword,
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
```
|
||||||
+8
@@ -0,0 +1,8 @@
|
|||||||
|
```typescript
|
||||||
|
for (const field of ["content_hash", "url", "section_url"]) {
|
||||||
|
await client.createPayloadIndex(COLLECTION, {
|
||||||
|
field_name: field,
|
||||||
|
field_schema: "keyword",
|
||||||
|
});
|
||||||
|
}
|
||||||
|
```
|
||||||
+12
@@ -0,0 +1,12 @@
|
|||||||
|
```csharp
|
||||||
|
Dictionary<string, Value> Payload(Chunk chunk, string? lastUpdated = null) => new()
|
||||||
|
{
|
||||||
|
["url"] = chunk.Url,
|
||||||
|
["anchor"] = chunk.Anchor,
|
||||||
|
["chunk_num"] = chunk.ChunkNum,
|
||||||
|
["section_url"] = chunk.SectionUrl,
|
||||||
|
["text"] = chunk.Text,
|
||||||
|
["content_hash"] = chunk.ContentHash,
|
||||||
|
["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
|
||||||
|
};
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```go
|
||||||
|
payload := func(c Chunk, lastUpdated string) map[string]any {
|
||||||
|
if lastUpdated == "" {
|
||||||
|
lastUpdated = time.Now().UTC().Format(time.RFC3339)
|
||||||
|
}
|
||||||
|
return map[string]any{
|
||||||
|
"url": c.URL,
|
||||||
|
"anchor": c.Anchor,
|
||||||
|
"chunk_num": c.ChunkNum,
|
||||||
|
"section_url": c.SectionURL,
|
||||||
|
"text": c.Text,
|
||||||
|
"content_hash": c.ContentHash,
|
||||||
|
"last_updated": lastUpdated,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```java
|
||||||
|
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
|
||||||
|
Map<String, Value> p = new HashMap<>();
|
||||||
|
p.put("url", value(chunk.url));
|
||||||
|
p.put("anchor", value(chunk.anchor));
|
||||||
|
p.put("chunk_num", value(chunk.chunkNum));
|
||||||
|
p.put("section_url", value(chunk.sectionUrl));
|
||||||
|
p.put("text", value(chunk.text));
|
||||||
|
p.put("content_hash", value(chunk.contentHash));
|
||||||
|
p.put("last_updated", value(lastUpdated != null
|
||||||
|
? lastUpdated
|
||||||
|
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
|
||||||
|
return p;
|
||||||
|
}
|
||||||
|
```
|
||||||
+12
@@ -0,0 +1,12 @@
|
|||||||
|
```python
|
||||||
|
def payload(chunk, last_updated=None):
|
||||||
|
return {
|
||||||
|
"url": chunk["url"],
|
||||||
|
"anchor": chunk["anchor"],
|
||||||
|
"chunk_num": chunk["chunk_num"],
|
||||||
|
"section_url": chunk["section_url"],
|
||||||
|
"text": chunk["text"],
|
||||||
|
"content_hash": chunk["content_hash"],
|
||||||
|
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||||
|
}
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```rust
|
||||||
|
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
|
||||||
|
let last_updated = last_updated.unwrap_or_else(|| {
|
||||||
|
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
|
||||||
|
});
|
||||||
|
Ok(Payload::try_from(serde_json::json!({
|
||||||
|
"url": chunk.url,
|
||||||
|
"anchor": chunk.anchor,
|
||||||
|
"chunk_num": chunk.chunk_num,
|
||||||
|
"section_url": chunk.section_url,
|
||||||
|
"text": chunk.text,
|
||||||
|
"content_hash": chunk.content_hash,
|
||||||
|
"last_updated": last_updated,
|
||||||
|
}))?)
|
||||||
|
}
|
||||||
|
```
|
||||||
+13
@@ -0,0 +1,13 @@
|
|||||||
|
```typescript
|
||||||
|
function payload(chunk: SyncChunk, lastUpdated?: string) {
|
||||||
|
return {
|
||||||
|
url: chunk.url,
|
||||||
|
anchor: chunk.anchor,
|
||||||
|
chunk_num: chunk.chunk_num,
|
||||||
|
section_url: chunk.section_url,
|
||||||
|
text: chunk.text,
|
||||||
|
content_hash: chunk.content_hash,
|
||||||
|
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
```
|
||||||
+12
@@ -0,0 +1,12 @@
|
|||||||
|
```csharp
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
}).ToList(),
|
||||||
|
wait: true
|
||||||
|
);
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```go
|
||||||
|
var points []*qdrant.PointStruct
|
||||||
|
for _, c := range prepareChunksForSync(CHUNKS) {
|
||||||
|
points = append(points, &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
```
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
```java
|
||||||
|
static void populate() throws Exception {
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (Chunk c : prepareChunksForSync(CHUNKS)) {
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
client.upsertAsync(COLLECTION, points).get();
|
||||||
|
}
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```python
|
||||||
|
client.upsert(COLLECTION, points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
for c in prepare_chunks_for_sync(CHUNKS)
|
||||||
|
], wait=True)
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```rust
|
||||||
|
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
Ok(PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
))
|
||||||
|
})
|
||||||
|
.collect::<anyhow::Result<_>>()?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```typescript
|
||||||
|
await client.upsert(COLLECTION, {
|
||||||
|
points: prepareChunksForSync(CHUNKS).map((c) => ({
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
})),
|
||||||
|
wait: true,
|
||||||
|
});
|
||||||
|
```
|
||||||
+203
@@ -0,0 +1,203 @@
|
|||||||
|
```python
|
||||||
|
import os
|
||||||
|
|
||||||
|
from qdrant_client import QdrantClient, models
|
||||||
|
|
||||||
|
QDRANT_URL = os.getenv("QDRANT_URL")
|
||||||
|
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
||||||
|
|
||||||
|
client = QdrantClient(
|
||||||
|
url=QDRANT_URL,
|
||||||
|
api_key=QDRANT_API_KEY,
|
||||||
|
cloud_inference=True
|
||||||
|
)
|
||||||
|
|
||||||
|
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
||||||
|
PIPELINE = "docs-prep-pipeline-v1"
|
||||||
|
COLLECTION = "docs-sync-tutorial"
|
||||||
|
|
||||||
|
client.create_collection(
|
||||||
|
COLLECTION,
|
||||||
|
vectors_config=models.VectorParams(
|
||||||
|
size=384, # all-MiniLM-L6-v2 output dimension
|
||||||
|
distance=models.Distance.COSINE,
|
||||||
|
),
|
||||||
|
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
||||||
|
)
|
||||||
|
|
||||||
|
def check_gate():
|
||||||
|
# compare this pipeline's constants against what the collection records about itself
|
||||||
|
meta = client.get_collection(COLLECTION).config.metadata or {}
|
||||||
|
|
||||||
|
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
||||||
|
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import uuid
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
def content_hash(text):
|
||||||
|
return hashlib.sha256(text.encode()).hexdigest()
|
||||||
|
|
||||||
|
def point_id(url, anchor, num):
|
||||||
|
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||||
|
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
||||||
|
|
||||||
|
def prepare_chunks_for_sync(chunks):
|
||||||
|
"""Derive both values (and the section address) for every raw chunk."""
|
||||||
|
out = []
|
||||||
|
for c in chunks:
|
||||||
|
text = normalize(c["text"])
|
||||||
|
out.append({
|
||||||
|
**c,
|
||||||
|
"text": text,
|
||||||
|
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
||||||
|
"content_hash": content_hash(text),
|
||||||
|
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
|
||||||
|
def payload(chunk, last_updated=None):
|
||||||
|
return {
|
||||||
|
"url": chunk["url"],
|
||||||
|
"anchor": chunk["anchor"],
|
||||||
|
"chunk_num": chunk["chunk_num"],
|
||||||
|
"section_url": chunk["section_url"],
|
||||||
|
"text": chunk["text"],
|
||||||
|
"content_hash": chunk["content_hash"],
|
||||||
|
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||||
|
}
|
||||||
|
|
||||||
|
for field in ("content_hash", "url", "section_url"):
|
||||||
|
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
||||||
|
|
||||||
|
client.upsert(COLLECTION, points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
for c in prepare_chunks_for_sync(CHUNKS)
|
||||||
|
], wait=True)
|
||||||
|
|
||||||
|
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||||
|
|
||||||
|
client.query_points(
|
||||||
|
COLLECTION,
|
||||||
|
query=models.Document(text=QUERY, model=MODEL),
|
||||||
|
limit=3,
|
||||||
|
with_payload=["section_url", "text"],
|
||||||
|
)
|
||||||
|
|
||||||
|
def split_by_state(latest_chunks):
|
||||||
|
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
||||||
|
incoming = {c["point_id"]: c for c in latest_chunks}
|
||||||
|
|
||||||
|
stored = {}
|
||||||
|
points = client.retrieve(
|
||||||
|
COLLECTION,
|
||||||
|
ids=list(incoming),
|
||||||
|
with_payload=["content_hash"],
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
for p in points:
|
||||||
|
stored[str(p.id)] = p.payload["content_hash"]
|
||||||
|
|
||||||
|
unchanged, content_changed, unknown_ids = [], [], []
|
||||||
|
for pid, c in incoming.items():
|
||||||
|
if stored.get(pid) == c["content_hash"]:
|
||||||
|
unchanged.append(c)
|
||||||
|
elif pid in stored:
|
||||||
|
content_changed.append(c)
|
||||||
|
else:
|
||||||
|
unknown_ids.append(c)
|
||||||
|
|
||||||
|
return incoming, unchanged, content_changed, unknown_ids
|
||||||
|
|
||||||
|
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
||||||
|
|
||||||
|
def re_embed_changed(content_changed):
|
||||||
|
if not content_changed:
|
||||||
|
return
|
||||||
|
client.upsert(COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
for c in content_changed],
|
||||||
|
wait=True)
|
||||||
|
|
||||||
|
def reuse_or_add(unknown_ids):
|
||||||
|
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
||||||
|
reused, added = 0, 0
|
||||||
|
|
||||||
|
for c in unknown_ids:
|
||||||
|
same_text = models.Filter(must=[
|
||||||
|
models.FieldCondition(
|
||||||
|
key="content_hash",
|
||||||
|
match=models.MatchValue(value=c["content_hash"]),
|
||||||
|
)
|
||||||
|
])
|
||||||
|
hits, _ = client.scroll(
|
||||||
|
COLLECTION,
|
||||||
|
scroll_filter=same_text,
|
||||||
|
limit=1,
|
||||||
|
with_payload=["last_updated"],
|
||||||
|
with_vectors=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if hits: # same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=hits[0].vector,
|
||||||
|
payload=payload(c, hits[0].payload["last_updated"]),
|
||||||
|
)
|
||||||
|
reused += 1
|
||||||
|
else: # genuinely new content: embed and insert
|
||||||
|
point = models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
added += 1
|
||||||
|
|
||||||
|
client.upsert(COLLECTION, points=[point], wait=True)
|
||||||
|
|
||||||
|
return reused, added
|
||||||
|
|
||||||
|
def delete_gone(incoming_ids):
|
||||||
|
"""Remove every point the current crawl no longer contains. Returns how many."""
|
||||||
|
if not incoming_ids:
|
||||||
|
raise ValueError("Refusing to delete from an empty source snapshot.")
|
||||||
|
|
||||||
|
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
||||||
|
|
||||||
|
to_delete = client.count(COLLECTION, count_filter=stale).count
|
||||||
|
|
||||||
|
# potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
||||||
|
return to_delete
|
||||||
|
|
||||||
|
def sync(latest_chunks):
|
||||||
|
check_gate() # refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
chunks = prepare_chunks_for_sync(latest_chunks)
|
||||||
|
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
||||||
|
|
||||||
|
re_embed_changed(content_changed)
|
||||||
|
reused, added = reuse_or_add(unknown_ids)
|
||||||
|
deleted = delete_gone(incoming_ids)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"unchanged": len(unchanged),
|
||||||
|
"re-embedded": len(content_changed),
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
}
|
||||||
|
|
||||||
|
run = sync(LATEST_CHUNKS)
|
||||||
|
print(run)
|
||||||
|
```
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
```csharp
|
||||||
|
async Task ReEmbedChanged(List<Chunk> contentChanged)
|
||||||
|
{
|
||||||
|
if (contentChanged.Count == 0)
|
||||||
|
return;
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
points: contentChanged.Select(c => new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
}).ToList(),
|
||||||
|
wait: true
|
||||||
|
);
|
||||||
|
}
|
||||||
|
```
|
||||||
+20
@@ -0,0 +1,20 @@
|
|||||||
|
```go
|
||||||
|
reEmbedChanged := func(contentChanged []Chunk) {
|
||||||
|
if len(contentChanged) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
|
||||||
|
for _, c := range contentChanged {
|
||||||
|
points = append(points, &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
```
|
||||||
+23
@@ -0,0 +1,23 @@
|
|||||||
|
```java
|
||||||
|
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
|
||||||
|
if (contentChanged.isEmpty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (Chunk c : contentChanged) {
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
client.upsertAsync(COLLECTION, points).get();
|
||||||
|
}
|
||||||
|
```
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
```python
|
||||||
|
def re_embed_changed(content_changed):
|
||||||
|
if not content_changed:
|
||||||
|
return
|
||||||
|
client.upsert(COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
for c in content_changed],
|
||||||
|
wait=True)
|
||||||
|
```
|
||||||
+22
@@ -0,0 +1,22 @@
|
|||||||
|
```rust
|
||||||
|
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
|
||||||
|
if content_changed.is_empty() {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let points: Vec<PointStruct> = content_changed
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
Ok(PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
))
|
||||||
|
})
|
||||||
|
.collect::<anyhow::Result<_>>()?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```typescript
|
||||||
|
async function reEmbedChanged(contentChanged: SyncChunk[]) {
|
||||||
|
if (contentChanged.length === 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
await client.upsert(COLLECTION, {
|
||||||
|
points: contentChanged.map((c) => ({
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
})),
|
||||||
|
wait: true,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
```
|
||||||
+48
@@ -0,0 +1,48 @@
|
|||||||
|
```csharp
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async Task<(int reused, int added)> ReuseOrAdd(List<Chunk> unknownIds)
|
||||||
|
{
|
||||||
|
int reused = 0, added = 0;
|
||||||
|
|
||||||
|
foreach (var c in unknownIds)
|
||||||
|
{
|
||||||
|
var sameText = new Filter
|
||||||
|
{
|
||||||
|
Must = { MatchKeyword("content_hash", c.ContentHash) }
|
||||||
|
};
|
||||||
|
var hits = (await client.ScrollAsync(
|
||||||
|
COLLECTION,
|
||||||
|
filter: sameText,
|
||||||
|
limit: 1,
|
||||||
|
payloadSelector: new[] { "last_updated" },
|
||||||
|
vectorsSelector: true
|
||||||
|
)).Result;
|
||||||
|
|
||||||
|
PointStruct point;
|
||||||
|
if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
|
||||||
|
{
|
||||||
|
point = new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
|
||||||
|
Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
|
||||||
|
};
|
||||||
|
reused++;
|
||||||
|
}
|
||||||
|
else // genuinely new content: embed and insert
|
||||||
|
{
|
||||||
|
point = new PointStruct
|
||||||
|
{
|
||||||
|
Id = new PointId { Uuid = c.PointId },
|
||||||
|
Vectors = new Document { Text = c.Text, Model = MODEL },
|
||||||
|
Payload = { Payload(c) },
|
||||||
|
};
|
||||||
|
added++;
|
||||||
|
}
|
||||||
|
|
||||||
|
await client.UpsertAsync(COLLECTION, points: new List<PointStruct> { point }, wait: true);
|
||||||
|
}
|
||||||
|
|
||||||
|
return (reused, added);
|
||||||
|
}
|
||||||
|
```
|
||||||
+46
@@ -0,0 +1,46 @@
|
|||||||
|
```go
|
||||||
|
// reuse an existing embedding when the same text is already stored; embed only what is new
|
||||||
|
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
|
||||||
|
reused, added := 0, 0
|
||||||
|
|
||||||
|
for _, c := range unknownIDs {
|
||||||
|
sameText := &qdrant.Filter{
|
||||||
|
Must: []*qdrant.Condition{
|
||||||
|
qdrant.NewMatch("content_hash", c.ContentHash),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Filter: sameText,
|
||||||
|
Limit: qdrant.PtrOf(uint32(1)),
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
|
||||||
|
WithVectors: qdrant.NewWithVectors(true),
|
||||||
|
})
|
||||||
|
|
||||||
|
var point *qdrant.PointStruct
|
||||||
|
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
|
||||||
|
}
|
||||||
|
reused++
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
}
|
||||||
|
added++
|
||||||
|
}
|
||||||
|
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{point},
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
return reused, added
|
||||||
|
}
|
||||||
|
```
|
||||||
+52
@@ -0,0 +1,52 @@
|
|||||||
|
```java
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
|
||||||
|
int reused = 0;
|
||||||
|
int added = 0;
|
||||||
|
|
||||||
|
for (Chunk c : unknownIds) {
|
||||||
|
Filter sameText = Filter.newBuilder()
|
||||||
|
.addMust(matchKeyword("content_hash", c.contentHash))
|
||||||
|
.build();
|
||||||
|
|
||||||
|
var hits = client.scrollAsync(
|
||||||
|
ScrollPoints.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setFilter(sameText)
|
||||||
|
.setLimit(1)
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
|
||||||
|
.setWithVectors(WithVectorsSelectorFactory.enable(true))
|
||||||
|
.build()).get().getResultList();
|
||||||
|
|
||||||
|
PointStruct point;
|
||||||
|
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(vectors(vector(
|
||||||
|
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
|
||||||
|
.getDataList())))
|
||||||
|
.putAllPayload(
|
||||||
|
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
|
||||||
|
.build();
|
||||||
|
reused++;
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build();
|
||||||
|
added++;
|
||||||
|
}
|
||||||
|
|
||||||
|
client.upsertAsync(COLLECTION, List.of(point)).get();
|
||||||
|
}
|
||||||
|
|
||||||
|
return new int[] {reused, added};
|
||||||
|
}
|
||||||
|
```
|
||||||
+39
@@ -0,0 +1,39 @@
|
|||||||
|
```python
|
||||||
|
def reuse_or_add(unknown_ids):
|
||||||
|
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
||||||
|
reused, added = 0, 0
|
||||||
|
|
||||||
|
for c in unknown_ids:
|
||||||
|
same_text = models.Filter(must=[
|
||||||
|
models.FieldCondition(
|
||||||
|
key="content_hash",
|
||||||
|
match=models.MatchValue(value=c["content_hash"]),
|
||||||
|
)
|
||||||
|
])
|
||||||
|
hits, _ = client.scroll(
|
||||||
|
COLLECTION,
|
||||||
|
scroll_filter=same_text,
|
||||||
|
limit=1,
|
||||||
|
with_payload=["last_updated"],
|
||||||
|
with_vectors=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if hits: # same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=hits[0].vector,
|
||||||
|
payload=payload(c, hits[0].payload["last_updated"]),
|
||||||
|
)
|
||||||
|
reused += 1
|
||||||
|
else: # genuinely new content: embed and insert
|
||||||
|
point = models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
added += 1
|
||||||
|
|
||||||
|
client.upsert(COLLECTION, points=[point], wait=True)
|
||||||
|
|
||||||
|
return reused, added
|
||||||
|
```
|
||||||
+51
@@ -0,0 +1,51 @@
|
|||||||
|
```rust
|
||||||
|
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
|
||||||
|
let (mut reused, mut added) = (0, 0);
|
||||||
|
|
||||||
|
for c in unknown_ids {
|
||||||
|
let same_text =
|
||||||
|
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
|
||||||
|
let hits = client
|
||||||
|
.scroll(
|
||||||
|
ScrollPointsBuilder::new(COLLECTION)
|
||||||
|
.filter(same_text)
|
||||||
|
.limit(1)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec![
|
||||||
|
"last_updated".to_string()
|
||||||
|
]))
|
||||||
|
.with_vectors(true),
|
||||||
|
)
|
||||||
|
.await?
|
||||||
|
.result;
|
||||||
|
|
||||||
|
let point = if let Some(hit) = hits.into_iter().next() {
|
||||||
|
// same text, new address: copy the vector, keep its last_updated
|
||||||
|
let last_updated = hit.get("last_updated").as_str().cloned();
|
||||||
|
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
|
||||||
|
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
|
||||||
|
Some(vector_output::Vector::Dense(dense)) => dense.data,
|
||||||
|
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||||
|
},
|
||||||
|
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||||
|
};
|
||||||
|
reused += 1;
|
||||||
|
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
|
||||||
|
} else {
|
||||||
|
// genuinely new content: embed and insert
|
||||||
|
added += 1;
|
||||||
|
PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((reused, added))
|
||||||
|
}
|
||||||
|
```
|
||||||
+45
@@ -0,0 +1,45 @@
|
|||||||
|
```typescript
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async function reuseOrAdd(unknownIds: SyncChunk[]) {
|
||||||
|
let reused = 0;
|
||||||
|
let added = 0;
|
||||||
|
|
||||||
|
for (const c of unknownIds) {
|
||||||
|
const sameText = {
|
||||||
|
must: [
|
||||||
|
{
|
||||||
|
key: "content_hash",
|
||||||
|
match: { value: c.content_hash },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
};
|
||||||
|
const hits = (await client.scroll(COLLECTION, {
|
||||||
|
filter: sameText,
|
||||||
|
limit: 1,
|
||||||
|
with_payload: ["last_updated"],
|
||||||
|
with_vector: true,
|
||||||
|
})).points;
|
||||||
|
|
||||||
|
let point: Schemas["PointStruct"];
|
||||||
|
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = {
|
||||||
|
id: c.point_id,
|
||||||
|
vector: hits[0].vector as number[],
|
||||||
|
payload: payload(c, hits[0].payload?.last_updated as string),
|
||||||
|
};
|
||||||
|
reused += 1;
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = {
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
};
|
||||||
|
added += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
await client.upsert(COLLECTION, { points: [point], wait: true });
|
||||||
|
}
|
||||||
|
|
||||||
|
return { reused, added };
|
||||||
|
}
|
||||||
|
```
|
||||||
+5
@@ -0,0 +1,5 @@
|
|||||||
|
```csharp
|
||||||
|
var run = await Sync(LATEST_CHUNKS);
|
||||||
|
foreach (var (op, count) in run)
|
||||||
|
Console.WriteLine($"{op}: {count}");
|
||||||
|
```
|
||||||
+4
@@ -0,0 +1,4 @@
|
|||||||
|
```go
|
||||||
|
run := sync(LATEST_CHUNKS)
|
||||||
|
fmt.Println(run)
|
||||||
|
```
|
||||||
+6
@@ -0,0 +1,6 @@
|
|||||||
|
```java
|
||||||
|
static void runSync() throws Exception {
|
||||||
|
Map<String, Long> run = sync(LATEST_CHUNKS);
|
||||||
|
System.out.println(run);
|
||||||
|
}
|
||||||
|
```
|
||||||
+4
@@ -0,0 +1,4 @@
|
|||||||
|
```python
|
||||||
|
run = sync(LATEST_CHUNKS)
|
||||||
|
print(run)
|
||||||
|
```
|
||||||
+4
@@ -0,0 +1,4 @@
|
|||||||
|
```rust
|
||||||
|
let run = sync(&client, &latest_chunks).await?;
|
||||||
|
println!("{run:?}");
|
||||||
|
```
|
||||||
+4
@@ -0,0 +1,4 @@
|
|||||||
|
```typescript
|
||||||
|
const run = await sync(LATEST_CHUNKS);
|
||||||
|
console.log(run);
|
||||||
|
```
|
||||||
+322
@@ -0,0 +1,322 @@
|
|||||||
|
```rust
|
||||||
|
use serde_json::{json, Value};
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use qdrant_client::qdrant::{
|
||||||
|
point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
|
||||||
|
CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
|
||||||
|
Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
|
||||||
|
Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
|
||||||
|
};
|
||||||
|
use qdrant_client::{Payload, Qdrant};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
let qdrant_url = std::env::var("QDRANT_URL")?;
|
||||||
|
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
|
||||||
|
|
||||||
|
let client = Qdrant::from_url(&qdrant_url)
|
||||||
|
.api_key(qdrant_api_key)
|
||||||
|
.build()?;
|
||||||
|
|
||||||
|
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
const PIPELINE: &str = "docs-prep-pipeline-v1";
|
||||||
|
const COLLECTION: &str = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
let mut metadata: HashMap<String, Value> = HashMap::new();
|
||||||
|
metadata.insert("embedding_model".to_string(), json!(MODEL));
|
||||||
|
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
|
||||||
|
|
||||||
|
client
|
||||||
|
.create_collection(
|
||||||
|
CreateCollectionBuilder::new(COLLECTION)
|
||||||
|
.vectors_config(VectorParamsBuilder::new(
|
||||||
|
384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance::Cosine,
|
||||||
|
))
|
||||||
|
.metadata(metadata),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
let meta = client
|
||||||
|
.collection_info(COLLECTION)
|
||||||
|
.await?
|
||||||
|
.result
|
||||||
|
.and_then(|info| info.config)
|
||||||
|
.map(|config| config.metadata)
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|
||||||
|
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
|
||||||
|
!= Some(PIPELINE)
|
||||||
|
{
|
||||||
|
anyhow::bail!(
|
||||||
|
"collection was built by {meta:?}: full re-embed into a fresh collection required"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn content_hash(text: &str) -> String {
|
||||||
|
Sha256::digest(text.as_bytes())
|
||||||
|
.iter()
|
||||||
|
.map(|byte| format!("{byte:02x}"))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn point_id(url: &str, anchor: &str, num: u32) -> String {
|
||||||
|
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||||
|
uuid::Uuid::new_v5(
|
||||||
|
&uuid::Uuid::NAMESPACE_URL,
|
||||||
|
format!("{url}#{anchor}::{num}").as_bytes(),
|
||||||
|
)
|
||||||
|
.to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Derive both values (and the section address) for every raw chunk.
|
||||||
|
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
|
||||||
|
chunks
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
let text = normalize(&c.text);
|
||||||
|
Chunk {
|
||||||
|
text: text.clone(),
|
||||||
|
section_url: if c.anchor.is_empty() {
|
||||||
|
c.url.clone()
|
||||||
|
} else {
|
||||||
|
format!("{}#{}", c.url, c.anchor)
|
||||||
|
},
|
||||||
|
content_hash: content_hash(&text),
|
||||||
|
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
|
||||||
|
..c.clone()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
|
||||||
|
let last_updated = last_updated.unwrap_or_else(|| {
|
||||||
|
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
|
||||||
|
});
|
||||||
|
Ok(Payload::try_from(serde_json::json!({
|
||||||
|
"url": chunk.url,
|
||||||
|
"anchor": chunk.anchor,
|
||||||
|
"chunk_num": chunk.chunk_num,
|
||||||
|
"section_url": chunk.section_url,
|
||||||
|
"text": chunk.text,
|
||||||
|
"content_hash": chunk.content_hash,
|
||||||
|
"last_updated": last_updated,
|
||||||
|
}))?)
|
||||||
|
}
|
||||||
|
|
||||||
|
for field in ["content_hash", "url", "section_url"] {
|
||||||
|
client
|
||||||
|
.create_field_index(CreateFieldIndexCollectionBuilder::new(
|
||||||
|
COLLECTION,
|
||||||
|
field,
|
||||||
|
FieldType::Keyword,
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
Ok(PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
))
|
||||||
|
})
|
||||||
|
.collect::<anyhow::Result<_>>()?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(COLLECTION)
|
||||||
|
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
|
||||||
|
.limit(3)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec![
|
||||||
|
"section_url".to_string(),
|
||||||
|
"text".to_string(),
|
||||||
|
])),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async fn split_by_state(
|
||||||
|
client: &Qdrant,
|
||||||
|
latest_chunks: &[Chunk],
|
||||||
|
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
|
||||||
|
let incoming: HashMap<String, Chunk> = latest_chunks
|
||||||
|
.iter()
|
||||||
|
.map(|c| (c.point_id.clone(), c.clone()))
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
|
||||||
|
let points = client
|
||||||
|
.get_points(
|
||||||
|
GetPointsBuilder::new(COLLECTION, ids)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
|
||||||
|
.with_vectors(false),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let mut stored: HashMap<String, String> = HashMap::new();
|
||||||
|
for p in points.result {
|
||||||
|
let hash = p.get("content_hash").as_str().cloned();
|
||||||
|
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
|
||||||
|
(p.id.and_then(|i| i.point_id_options), hash)
|
||||||
|
{
|
||||||
|
stored.insert(id, hash);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let (mut unchanged, mut content_changed, mut unknown_ids) =
|
||||||
|
(Vec::new(), Vec::new(), Vec::new());
|
||||||
|
for (pid, c) in &incoming {
|
||||||
|
if stored.get(pid) == Some(&c.content_hash) {
|
||||||
|
unchanged.push(c.clone());
|
||||||
|
} else if stored.contains_key(pid) {
|
||||||
|
content_changed.push(c.clone());
|
||||||
|
} else {
|
||||||
|
unknown_ids.push(c.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((incoming, unchanged, content_changed, unknown_ids))
|
||||||
|
}
|
||||||
|
|
||||||
|
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||||
|
split_by_state(&client, &latest_chunks).await?;
|
||||||
|
|
||||||
|
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
|
||||||
|
if content_changed.is_empty() {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let points: Vec<PointStruct> = content_changed
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
Ok(PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
))
|
||||||
|
})
|
||||||
|
.collect::<anyhow::Result<_>>()?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
|
||||||
|
let (mut reused, mut added) = (0, 0);
|
||||||
|
|
||||||
|
for c in unknown_ids {
|
||||||
|
let same_text =
|
||||||
|
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
|
||||||
|
let hits = client
|
||||||
|
.scroll(
|
||||||
|
ScrollPointsBuilder::new(COLLECTION)
|
||||||
|
.filter(same_text)
|
||||||
|
.limit(1)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec![
|
||||||
|
"last_updated".to_string()
|
||||||
|
]))
|
||||||
|
.with_vectors(true),
|
||||||
|
)
|
||||||
|
.await?
|
||||||
|
.result;
|
||||||
|
|
||||||
|
let point = if let Some(hit) = hits.into_iter().next() {
|
||||||
|
// same text, new address: copy the vector, keep its last_updated
|
||||||
|
let last_updated = hit.get("last_updated").as_str().cloned();
|
||||||
|
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
|
||||||
|
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
|
||||||
|
Some(vector_output::Vector::Dense(dense)) => dense.data,
|
||||||
|
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||||
|
},
|
||||||
|
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||||
|
};
|
||||||
|
reused += 1;
|
||||||
|
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
|
||||||
|
} else {
|
||||||
|
// genuinely new content: embed and insert
|
||||||
|
added += 1;
|
||||||
|
PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((reused, added))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async fn delete_gone(
|
||||||
|
client: &Qdrant,
|
||||||
|
incoming_ids: &HashMap<String, Chunk>,
|
||||||
|
) -> anyhow::Result<u64> {
|
||||||
|
if incoming_ids.is_empty() {
|
||||||
|
anyhow::bail!("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
let stale = Filter::must_not([Condition::has_id(
|
||||||
|
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
|
||||||
|
)]);
|
||||||
|
|
||||||
|
let to_delete = client
|
||||||
|
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
|
||||||
|
.await?
|
||||||
|
.result
|
||||||
|
.map(|r| r.count)
|
||||||
|
.unwrap_or(0);
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client
|
||||||
|
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
|
||||||
|
.await?;
|
||||||
|
Ok(to_delete)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn sync(
|
||||||
|
client: &Qdrant,
|
||||||
|
latest_chunks: &[Chunk],
|
||||||
|
) -> anyhow::Result<HashMap<&'static str, usize>> {
|
||||||
|
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
let chunks = prepare_chunks_for_sync(latest_chunks);
|
||||||
|
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||||
|
split_by_state(client, &chunks).await?;
|
||||||
|
|
||||||
|
re_embed_changed(client, &content_changed).await?;
|
||||||
|
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
|
||||||
|
let deleted = delete_gone(client, &incoming_ids).await?;
|
||||||
|
|
||||||
|
Ok(HashMap::from([
|
||||||
|
("unchanged", unchanged.len()),
|
||||||
|
("re-embedded", content_changed.len()),
|
||||||
|
("reused_embedding", reused),
|
||||||
|
("added", added),
|
||||||
|
("deleted", deleted as usize),
|
||||||
|
]))
|
||||||
|
}
|
||||||
|
|
||||||
|
let run = sync(&client, &latest_chunks).await?;
|
||||||
|
println!("{run:?}");
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```csharp
|
||||||
|
var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
await client.QueryAsync(
|
||||||
|
collectionName: COLLECTION,
|
||||||
|
query: new Document { Text = QUERY, Model = MODEL },
|
||||||
|
limit: 3,
|
||||||
|
payloadSelector: new[] { "section_url", "text" }
|
||||||
|
);
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```go
|
||||||
|
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||||
|
|
||||||
|
client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(3)),
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
|
||||||
|
})
|
||||||
|
```
|
||||||
+19
@@ -0,0 +1,19 @@
|
|||||||
|
```java
|
||||||
|
static final String QUERY =
|
||||||
|
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
static void search() throws Exception {
|
||||||
|
client.queryAsync(
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(QUERY)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(3)
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
|
||||||
|
.build()).get();
|
||||||
|
}
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```python
|
||||||
|
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||||
|
|
||||||
|
client.query_points(
|
||||||
|
COLLECTION,
|
||||||
|
query=models.Document(text=QUERY, model=MODEL),
|
||||||
|
limit=3,
|
||||||
|
with_payload=["section_url", "text"],
|
||||||
|
)
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```rust
|
||||||
|
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(COLLECTION)
|
||||||
|
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
|
||||||
|
.limit(3)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec![
|
||||||
|
"section_url".to_string(),
|
||||||
|
"text".to_string(),
|
||||||
|
])),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+9
@@ -0,0 +1,9 @@
|
|||||||
|
```typescript
|
||||||
|
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
await client.query(COLLECTION, {
|
||||||
|
query: { text: QUERY, model: MODEL },
|
||||||
|
limit: 3,
|
||||||
|
with_payload: ["section_url", "text"],
|
||||||
|
});
|
||||||
|
```
|
||||||
+35
@@ -0,0 +1,35 @@
|
|||||||
|
```csharp
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async Task<(Dictionary<string, Chunk> incomingIds, List<Chunk> unchanged, List<Chunk> contentChanged, List<Chunk> unknownIds)>
|
||||||
|
SplitByState(List<Chunk> latestChunks)
|
||||||
|
{
|
||||||
|
var incoming = latestChunks.ToDictionary(c => c.PointId);
|
||||||
|
|
||||||
|
var stored = new Dictionary<string, string>();
|
||||||
|
var points = await client.RetrieveAsync(
|
||||||
|
COLLECTION,
|
||||||
|
ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
|
||||||
|
payloadSelector: new[] { "content_hash" },
|
||||||
|
vectorSelector: false
|
||||||
|
);
|
||||||
|
foreach (var p in points)
|
||||||
|
stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
|
||||||
|
|
||||||
|
var unchanged = new List<Chunk>();
|
||||||
|
var contentChanged = new List<Chunk>();
|
||||||
|
var unknownIds = new List<Chunk>();
|
||||||
|
foreach (var (pid, c) in incoming)
|
||||||
|
{
|
||||||
|
if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
|
||||||
|
unchanged.Add(c);
|
||||||
|
else if (stored.ContainsKey(pid))
|
||||||
|
contentChanged.Add(c);
|
||||||
|
else
|
||||||
|
unknownIds.Add(c);
|
||||||
|
}
|
||||||
|
|
||||||
|
return (incoming, unchanged, contentChanged, unknownIds);
|
||||||
|
}
|
||||||
|
|
||||||
|
var splitState = await SplitByState(LATEST_CHUNKS);
|
||||||
|
```
|
||||||
+39
@@ -0,0 +1,39 @@
|
|||||||
|
```go
|
||||||
|
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
|
||||||
|
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
|
||||||
|
incoming := make(map[string]Chunk, len(latestChunks))
|
||||||
|
ids := make([]*qdrant.PointId, 0, len(latestChunks))
|
||||||
|
for _, c := range latestChunks {
|
||||||
|
incoming[c.PointID] = c
|
||||||
|
ids = append(ids, qdrant.NewID(c.PointID))
|
||||||
|
}
|
||||||
|
|
||||||
|
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Ids: ids,
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
|
||||||
|
WithVectors: qdrant.NewWithVectors(false),
|
||||||
|
})
|
||||||
|
stored := make(map[string]string, len(retrieved))
|
||||||
|
for _, p := range retrieved {
|
||||||
|
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
|
||||||
|
}
|
||||||
|
|
||||||
|
var unchanged, contentChanged, unknownIDs []Chunk
|
||||||
|
for pid, c := range incoming {
|
||||||
|
storedHash, found := stored[pid]
|
||||||
|
switch {
|
||||||
|
case found && storedHash == c.ContentHash:
|
||||||
|
unchanged = append(unchanged, c)
|
||||||
|
case found:
|
||||||
|
contentChanged = append(contentChanged, c)
|
||||||
|
default:
|
||||||
|
unknownIDs = append(unknownIDs, c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return incoming, unchanged, contentChanged, unknownIDs
|
||||||
|
}
|
||||||
|
|
||||||
|
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
|
||||||
|
```
|
||||||
+43
@@ -0,0 +1,43 @@
|
|||||||
|
```java
|
||||||
|
static class SyncState {
|
||||||
|
Map<String, Chunk> incoming = new LinkedHashMap<>();
|
||||||
|
List<Chunk> unchanged = new ArrayList<>();
|
||||||
|
List<Chunk> contentChanged = new ArrayList<>();
|
||||||
|
List<Chunk> unknownIds = new ArrayList<>();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
|
||||||
|
SyncState state = new SyncState();
|
||||||
|
for (Chunk c : latestChunks) {
|
||||||
|
state.incoming.put(c.pointId, c);
|
||||||
|
}
|
||||||
|
|
||||||
|
Map<String, String> stored = new HashMap<>();
|
||||||
|
var points = client.retrieveAsync(
|
||||||
|
COLLECTION,
|
||||||
|
state.incoming.keySet().stream()
|
||||||
|
.map(pid -> id(UUID.fromString(pid)))
|
||||||
|
.collect(Collectors.toList()),
|
||||||
|
WithPayloadSelectorFactory.include(List.of("content_hash")),
|
||||||
|
WithVectorsSelectorFactory.enable(false),
|
||||||
|
null).get();
|
||||||
|
for (var p : points) {
|
||||||
|
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
|
||||||
|
}
|
||||||
|
|
||||||
|
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
|
||||||
|
String pid = e.getKey();
|
||||||
|
Chunk c = e.getValue();
|
||||||
|
if (c.contentHash.equals(stored.get(pid))) {
|
||||||
|
state.unchanged.add(c);
|
||||||
|
} else if (stored.containsKey(pid)) {
|
||||||
|
state.contentChanged.add(c);
|
||||||
|
} else {
|
||||||
|
state.unknownIds.add(c);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return state;
|
||||||
|
}
|
||||||
|
```
|
||||||
+28
@@ -0,0 +1,28 @@
|
|||||||
|
```python
|
||||||
|
def split_by_state(latest_chunks):
|
||||||
|
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
||||||
|
incoming = {c["point_id"]: c for c in latest_chunks}
|
||||||
|
|
||||||
|
stored = {}
|
||||||
|
points = client.retrieve(
|
||||||
|
COLLECTION,
|
||||||
|
ids=list(incoming),
|
||||||
|
with_payload=["content_hash"],
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
for p in points:
|
||||||
|
stored[str(p.id)] = p.payload["content_hash"]
|
||||||
|
|
||||||
|
unchanged, content_changed, unknown_ids = [], [], []
|
||||||
|
for pid, c in incoming.items():
|
||||||
|
if stored.get(pid) == c["content_hash"]:
|
||||||
|
unchanged.append(c)
|
||||||
|
elif pid in stored:
|
||||||
|
content_changed.append(c)
|
||||||
|
else:
|
||||||
|
unknown_ids.append(c)
|
||||||
|
|
||||||
|
return incoming, unchanged, content_changed, unknown_ids
|
||||||
|
|
||||||
|
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
||||||
|
```
|
||||||
+48
@@ -0,0 +1,48 @@
|
|||||||
|
```rust
|
||||||
|
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async fn split_by_state(
|
||||||
|
client: &Qdrant,
|
||||||
|
latest_chunks: &[Chunk],
|
||||||
|
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
|
||||||
|
let incoming: HashMap<String, Chunk> = latest_chunks
|
||||||
|
.iter()
|
||||||
|
.map(|c| (c.point_id.clone(), c.clone()))
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
|
||||||
|
let points = client
|
||||||
|
.get_points(
|
||||||
|
GetPointsBuilder::new(COLLECTION, ids)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
|
||||||
|
.with_vectors(false),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let mut stored: HashMap<String, String> = HashMap::new();
|
||||||
|
for p in points.result {
|
||||||
|
let hash = p.get("content_hash").as_str().cloned();
|
||||||
|
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
|
||||||
|
(p.id.and_then(|i| i.point_id_options), hash)
|
||||||
|
{
|
||||||
|
stored.insert(id, hash);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let (mut unchanged, mut content_changed, mut unknown_ids) =
|
||||||
|
(Vec::new(), Vec::new(), Vec::new());
|
||||||
|
for (pid, c) in &incoming {
|
||||||
|
if stored.get(pid) == Some(&c.content_hash) {
|
||||||
|
unchanged.push(c.clone());
|
||||||
|
} else if stored.contains_key(pid) {
|
||||||
|
content_changed.push(c.clone());
|
||||||
|
} else {
|
||||||
|
unknown_ids.push(c.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((incoming, unchanged, content_changed, unknown_ids))
|
||||||
|
}
|
||||||
|
|
||||||
|
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||||
|
split_by_state(&client, &latest_chunks).await?;
|
||||||
|
```
|
||||||
+33
@@ -0,0 +1,33 @@
|
|||||||
|
```typescript
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async function splitByState(latestChunks: SyncChunk[]) {
|
||||||
|
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
|
||||||
|
|
||||||
|
const stored = new Map<string, string>();
|
||||||
|
const points = await client.retrieve(COLLECTION, {
|
||||||
|
ids: [...incoming.keys()],
|
||||||
|
with_payload: ["content_hash"],
|
||||||
|
with_vector: false,
|
||||||
|
});
|
||||||
|
for (const p of points) {
|
||||||
|
stored.set(String(p.id), p.payload?.content_hash as string);
|
||||||
|
}
|
||||||
|
|
||||||
|
const unchanged: SyncChunk[] = [];
|
||||||
|
const contentChanged: SyncChunk[] = [];
|
||||||
|
const unknownIds: SyncChunk[] = [];
|
||||||
|
for (const [pid, c] of incoming) {
|
||||||
|
if (stored.get(pid) === c.content_hash) {
|
||||||
|
unchanged.push(c);
|
||||||
|
} else if (stored.has(pid)) {
|
||||||
|
contentChanged.push(c);
|
||||||
|
} else {
|
||||||
|
unknownIds.push(c);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return { incoming, unchanged, contentChanged, unknownIds };
|
||||||
|
}
|
||||||
|
|
||||||
|
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
|
||||||
|
```
|
||||||
+22
@@ -0,0 +1,22 @@
|
|||||||
|
```csharp
|
||||||
|
async Task<Dictionary<string, long>> Sync(List<Chunk> latestChunks)
|
||||||
|
{
|
||||||
|
await CheckGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
var chunks = PrepareChunksForSync(latestChunks);
|
||||||
|
var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
|
||||||
|
|
||||||
|
await ReEmbedChanged(contentChanged);
|
||||||
|
var (reused, added) = await ReuseOrAdd(unknownIds);
|
||||||
|
var deleted = await DeleteGone(incomingIds);
|
||||||
|
|
||||||
|
return new Dictionary<string, long>
|
||||||
|
{
|
||||||
|
["unchanged"] = unchanged.Count,
|
||||||
|
["re-embedded"] = contentChanged.Count,
|
||||||
|
["reused_embedding"] = reused,
|
||||||
|
["added"] = added,
|
||||||
|
["deleted"] = (long)deleted,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
```
|
||||||
+20
@@ -0,0 +1,20 @@
|
|||||||
|
```go
|
||||||
|
sync := func(latestChunks []Chunk) map[string]int {
|
||||||
|
checkGate() // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
chunks := prepareChunksForSync(latestChunks)
|
||||||
|
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
|
||||||
|
|
||||||
|
reEmbedChanged(contentChanged)
|
||||||
|
reused, added := reuseOrAdd(unknownIDs)
|
||||||
|
deleted := deleteGone(incomingIDs)
|
||||||
|
|
||||||
|
return map[string]int{
|
||||||
|
"unchanged": len(unchanged),
|
||||||
|
"re-embedded": len(contentChanged),
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+19
@@ -0,0 +1,19 @@
|
|||||||
|
```java
|
||||||
|
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
|
||||||
|
checkGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
List<Chunk> chunks = prepareChunksForSync(latestChunks);
|
||||||
|
SyncState state = splitByState(chunks);
|
||||||
|
|
||||||
|
reEmbedChanged(state.contentChanged);
|
||||||
|
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
|
||||||
|
long deleted = deleteGone(state.incoming);
|
||||||
|
|
||||||
|
return Map.of(
|
||||||
|
"unchanged", (long) state.unchanged.size(),
|
||||||
|
"re-embedded", (long) state.contentChanged.size(),
|
||||||
|
"reused_embedding", (long) reusedAdded[0],
|
||||||
|
"added", (long) reusedAdded[1],
|
||||||
|
"deleted", deleted);
|
||||||
|
}
|
||||||
|
```
|
||||||
+19
@@ -0,0 +1,19 @@
|
|||||||
|
```python
|
||||||
|
def sync(latest_chunks):
|
||||||
|
check_gate() # refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
chunks = prepare_chunks_for_sync(latest_chunks)
|
||||||
|
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
||||||
|
|
||||||
|
re_embed_changed(content_changed)
|
||||||
|
reused, added = reuse_or_add(unknown_ids)
|
||||||
|
deleted = delete_gone(incoming_ids)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"unchanged": len(unchanged),
|
||||||
|
"re-embedded": len(content_changed),
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
}
|
||||||
|
```
|
||||||
+24
@@ -0,0 +1,24 @@
|
|||||||
|
```rust
|
||||||
|
async fn sync(
|
||||||
|
client: &Qdrant,
|
||||||
|
latest_chunks: &[Chunk],
|
||||||
|
) -> anyhow::Result<HashMap<&'static str, usize>> {
|
||||||
|
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
let chunks = prepare_chunks_for_sync(latest_chunks);
|
||||||
|
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||||
|
split_by_state(client, &chunks).await?;
|
||||||
|
|
||||||
|
re_embed_changed(client, &content_changed).await?;
|
||||||
|
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
|
||||||
|
let deleted = delete_gone(client, &incoming_ids).await?;
|
||||||
|
|
||||||
|
Ok(HashMap::from([
|
||||||
|
("unchanged", unchanged.len()),
|
||||||
|
("re-embedded", content_changed.len()),
|
||||||
|
("reused_embedding", reused),
|
||||||
|
("added", added),
|
||||||
|
("deleted", deleted as usize),
|
||||||
|
]))
|
||||||
|
}
|
||||||
|
```
|
||||||
+20
@@ -0,0 +1,20 @@
|
|||||||
|
```typescript
|
||||||
|
async function sync(latestChunks: RawChunk[]) {
|
||||||
|
await checkGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
const chunks = prepareChunksForSync(latestChunks);
|
||||||
|
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
|
||||||
|
|
||||||
|
await reEmbedChanged(contentChanged);
|
||||||
|
const { reused, added } = await reuseOrAdd(unknownIds);
|
||||||
|
const deleted = await deleteGone(incoming);
|
||||||
|
|
||||||
|
return {
|
||||||
|
"unchanged": unchanged.length,
|
||||||
|
"re-embedded": contentChanged.length,
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
```
|
||||||
+230
@@ -0,0 +1,230 @@
|
|||||||
|
```typescript
|
||||||
|
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
|
||||||
|
|
||||||
|
const QDRANT_URL = process.env.QDRANT_URL;
|
||||||
|
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
|
||||||
|
|
||||||
|
const client = new QdrantClient({
|
||||||
|
url: QDRANT_URL,
|
||||||
|
apiKey: QDRANT_API_KEY,
|
||||||
|
});
|
||||||
|
|
||||||
|
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
const PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
const COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
await client.createCollection(COLLECTION, {
|
||||||
|
vectors: {
|
||||||
|
size: 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
distance: "Cosine",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
await client.updateCollection(COLLECTION, {
|
||||||
|
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
|
||||||
|
});
|
||||||
|
|
||||||
|
async function checkGate() {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
|
||||||
|
{}) as Record<string, unknown>;
|
||||||
|
|
||||||
|
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
|
||||||
|
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
import { createHash } from "node:crypto";
|
||||||
|
|
||||||
|
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
|
||||||
|
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
|
||||||
|
|
||||||
|
function contentHash(text: string): string {
|
||||||
|
return createHash("sha256").update(text).digest("hex");
|
||||||
|
}
|
||||||
|
|
||||||
|
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
|
||||||
|
function pointId(url: string, anchor: string, num: number): string {
|
||||||
|
// Qdrant accepts any well-formed UUID as a point ID:
|
||||||
|
// hash the address, format the digest as a UUID, and the same address always yields the same ID
|
||||||
|
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
|
||||||
|
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
|
||||||
|
return chunks.map((c) => {
|
||||||
|
const text = normalize(c.text);
|
||||||
|
return {
|
||||||
|
...c,
|
||||||
|
text,
|
||||||
|
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
|
||||||
|
content_hash: contentHash(text),
|
||||||
|
point_id: pointId(c.url, c.anchor, c.chunk_num),
|
||||||
|
};
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
function payload(chunk: SyncChunk, lastUpdated?: string) {
|
||||||
|
return {
|
||||||
|
url: chunk.url,
|
||||||
|
anchor: chunk.anchor,
|
||||||
|
chunk_num: chunk.chunk_num,
|
||||||
|
section_url: chunk.section_url,
|
||||||
|
text: chunk.text,
|
||||||
|
content_hash: chunk.content_hash,
|
||||||
|
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const field of ["content_hash", "url", "section_url"]) {
|
||||||
|
await client.createPayloadIndex(COLLECTION, {
|
||||||
|
field_name: field,
|
||||||
|
field_schema: "keyword",
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
await client.upsert(COLLECTION, {
|
||||||
|
points: prepareChunksForSync(CHUNKS).map((c) => ({
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
})),
|
||||||
|
wait: true,
|
||||||
|
});
|
||||||
|
|
||||||
|
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
await client.query(COLLECTION, {
|
||||||
|
query: { text: QUERY, model: MODEL },
|
||||||
|
limit: 3,
|
||||||
|
with_payload: ["section_url", "text"],
|
||||||
|
});
|
||||||
|
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async function splitByState(latestChunks: SyncChunk[]) {
|
||||||
|
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
|
||||||
|
|
||||||
|
const stored = new Map<string, string>();
|
||||||
|
const points = await client.retrieve(COLLECTION, {
|
||||||
|
ids: [...incoming.keys()],
|
||||||
|
with_payload: ["content_hash"],
|
||||||
|
with_vector: false,
|
||||||
|
});
|
||||||
|
for (const p of points) {
|
||||||
|
stored.set(String(p.id), p.payload?.content_hash as string);
|
||||||
|
}
|
||||||
|
|
||||||
|
const unchanged: SyncChunk[] = [];
|
||||||
|
const contentChanged: SyncChunk[] = [];
|
||||||
|
const unknownIds: SyncChunk[] = [];
|
||||||
|
for (const [pid, c] of incoming) {
|
||||||
|
if (stored.get(pid) === c.content_hash) {
|
||||||
|
unchanged.push(c);
|
||||||
|
} else if (stored.has(pid)) {
|
||||||
|
contentChanged.push(c);
|
||||||
|
} else {
|
||||||
|
unknownIds.push(c);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return { incoming, unchanged, contentChanged, unknownIds };
|
||||||
|
}
|
||||||
|
|
||||||
|
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
|
||||||
|
|
||||||
|
async function reEmbedChanged(contentChanged: SyncChunk[]) {
|
||||||
|
if (contentChanged.length === 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
await client.upsert(COLLECTION, {
|
||||||
|
points: contentChanged.map((c) => ({
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
})),
|
||||||
|
wait: true,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async function reuseOrAdd(unknownIds: SyncChunk[]) {
|
||||||
|
let reused = 0;
|
||||||
|
let added = 0;
|
||||||
|
|
||||||
|
for (const c of unknownIds) {
|
||||||
|
const sameText = {
|
||||||
|
must: [
|
||||||
|
{
|
||||||
|
key: "content_hash",
|
||||||
|
match: { value: c.content_hash },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
};
|
||||||
|
const hits = (await client.scroll(COLLECTION, {
|
||||||
|
filter: sameText,
|
||||||
|
limit: 1,
|
||||||
|
with_payload: ["last_updated"],
|
||||||
|
with_vector: true,
|
||||||
|
})).points;
|
||||||
|
|
||||||
|
let point: Schemas["PointStruct"];
|
||||||
|
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = {
|
||||||
|
id: c.point_id,
|
||||||
|
vector: hits[0].vector as number[],
|
||||||
|
payload: payload(c, hits[0].payload?.last_updated as string),
|
||||||
|
};
|
||||||
|
reused += 1;
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = {
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
};
|
||||||
|
added += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
await client.upsert(COLLECTION, { points: [point], wait: true });
|
||||||
|
}
|
||||||
|
|
||||||
|
return { reused, added };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async function deleteGone(incoming: Map<string, SyncChunk>) {
|
||||||
|
if (incoming.size === 0) {
|
||||||
|
throw new Error("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
|
||||||
|
|
||||||
|
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
await client.delete(COLLECTION, { filter: stale, wait: true });
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function sync(latestChunks: RawChunk[]) {
|
||||||
|
await checkGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
const chunks = prepareChunksForSync(latestChunks);
|
||||||
|
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
|
||||||
|
|
||||||
|
await reEmbedChanged(contentChanged);
|
||||||
|
const { reused, added } = await reuseOrAdd(unknownIds);
|
||||||
|
const deleted = await deleteGone(incoming);
|
||||||
|
|
||||||
|
return {
|
||||||
|
"unchanged": unchanged.length,
|
||||||
|
"re-embedded": contentChanged.length,
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
const run = await sync(LATEST_CHUNKS);
|
||||||
|
console.log(run);
|
||||||
|
```
|
||||||
+359
@@ -0,0 +1,359 @@
|
|||||||
|
package snippet
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"crypto/sha256"
|
||||||
|
"encoding/hex"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/google/uuid"
|
||||||
|
"github.com/qdrant/go-client/qdrant"
|
||||||
|
)
|
||||||
|
|
||||||
|
func Main() {
|
||||||
|
// @block-start client-connection
|
||||||
|
QDRANT_URL := os.Getenv("QDRANT_URL")
|
||||||
|
QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
|
||||||
|
|
||||||
|
client, err := qdrant.NewClient(&qdrant.Config{
|
||||||
|
Host: QDRANT_URL,
|
||||||
|
APIKey: QDRANT_API_KEY,
|
||||||
|
UseTLS: true,
|
||||||
|
})
|
||||||
|
// @block-end client-connection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
if err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// data and text normalization are not the lesson of this tutorial:
|
||||||
|
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||||
|
type Chunk struct {
|
||||||
|
URL string
|
||||||
|
Anchor string
|
||||||
|
ChunkNum int
|
||||||
|
Text string
|
||||||
|
SectionURL string
|
||||||
|
ContentHash string
|
||||||
|
PointID string
|
||||||
|
}
|
||||||
|
|
||||||
|
CHUNKS := []Chunk{
|
||||||
|
{
|
||||||
|
URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
Anchor: "prerequisites",
|
||||||
|
ChunkNum: 0,
|
||||||
|
Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
Anchor: "step-3-enable-an-admin-api-key",
|
||||||
|
ChunkNum: 0,
|
||||||
|
Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
invisibleChars := regexp.MustCompile("[\u200B\u200C\u200D\uFEFF\u00AD]") // zero-width chars and soft hyphen
|
||||||
|
whitespace := regexp.MustCompile(`\s+`)
|
||||||
|
normalize := func(text string) string {
|
||||||
|
text = invisibleChars.ReplaceAllString(text, "")
|
||||||
|
return strings.TrimSpace(whitespace.ReplaceAllString(text, " "))
|
||||||
|
}
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-collection
|
||||||
|
MODEL := "sentence-transformers/all-MiniLM-L6-v2"
|
||||||
|
PIPELINE := "docs-prep-pipeline-v1"
|
||||||
|
COLLECTION := "docs-sync-tutorial"
|
||||||
|
|
||||||
|
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||||
|
Size: 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance: qdrant.Distance_Cosine,
|
||||||
|
}),
|
||||||
|
Metadata: qdrant.NewValueMap(map[string]any{
|
||||||
|
"embedding_model": MODEL,
|
||||||
|
"pipeline_version": PIPELINE,
|
||||||
|
}),
|
||||||
|
})
|
||||||
|
// @block-end create-collection
|
||||||
|
|
||||||
|
// @block-start check-gate
|
||||||
|
checkGate := func() {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
|
||||||
|
if err != nil { panic(err) } // @hide
|
||||||
|
meta := info.GetConfig().GetMetadata()
|
||||||
|
|
||||||
|
if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
|
||||||
|
panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end check-gate
|
||||||
|
|
||||||
|
// @block-start identity-and-fingerprint
|
||||||
|
contentHash := func(text string) string {
|
||||||
|
sum := sha256.Sum256([]byte(text))
|
||||||
|
return hex.EncodeToString(sum[:])
|
||||||
|
}
|
||||||
|
|
||||||
|
pointID := func(url, anchor string, num int) string {
|
||||||
|
// NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
|
||||||
|
// marking the input as a URL-like name
|
||||||
|
return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// derive both values (and the section address) for every raw chunk
|
||||||
|
prepareChunksForSync := func(chunks []Chunk) []Chunk {
|
||||||
|
out := make([]Chunk, 0, len(chunks))
|
||||||
|
for _, c := range chunks {
|
||||||
|
c.Text = normalize(c.Text)
|
||||||
|
c.SectionURL = c.URL
|
||||||
|
if c.Anchor != "" {
|
||||||
|
c.SectionURL = c.URL + "#" + c.Anchor
|
||||||
|
}
|
||||||
|
c.ContentHash = contentHash(c.Text)
|
||||||
|
c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
|
||||||
|
out = append(out, c)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
// @block-end identity-and-fingerprint
|
||||||
|
|
||||||
|
// @block-start payload
|
||||||
|
payload := func(c Chunk, lastUpdated string) map[string]any {
|
||||||
|
if lastUpdated == "" {
|
||||||
|
lastUpdated = time.Now().UTC().Format(time.RFC3339)
|
||||||
|
}
|
||||||
|
return map[string]any{
|
||||||
|
"url": c.URL,
|
||||||
|
"anchor": c.Anchor,
|
||||||
|
"chunk_num": c.ChunkNum,
|
||||||
|
"section_url": c.SectionURL,
|
||||||
|
"text": c.Text,
|
||||||
|
"content_hash": c.ContentHash,
|
||||||
|
"last_updated": lastUpdated,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end payload
|
||||||
|
|
||||||
|
// @block-start payload-indexes
|
||||||
|
for _, field := range []string{"content_hash", "url", "section_url"} {
|
||||||
|
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
FieldName: field,
|
||||||
|
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// @block-end payload-indexes
|
||||||
|
|
||||||
|
// @block-start populate
|
||||||
|
var points []*qdrant.PointStruct
|
||||||
|
for _, c := range prepareChunksForSync(CHUNKS) {
|
||||||
|
points = append(points, &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
// @block-end populate
|
||||||
|
|
||||||
|
// @block-start search
|
||||||
|
QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||||
|
|
||||||
|
client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(3)),
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
|
||||||
|
})
|
||||||
|
// @block-end search
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||||
|
LATEST_CHUNKS := prepareChunksForSync(CHUNKS)
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start split-by-state
|
||||||
|
// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
|
||||||
|
splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
|
||||||
|
incoming := make(map[string]Chunk, len(latestChunks))
|
||||||
|
ids := make([]*qdrant.PointId, 0, len(latestChunks))
|
||||||
|
for _, c := range latestChunks {
|
||||||
|
incoming[c.PointID] = c
|
||||||
|
ids = append(ids, qdrant.NewID(c.PointID))
|
||||||
|
}
|
||||||
|
|
||||||
|
retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Ids: ids,
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
|
||||||
|
WithVectors: qdrant.NewWithVectors(false),
|
||||||
|
})
|
||||||
|
if err != nil { panic(err) } // @hide
|
||||||
|
stored := make(map[string]string, len(retrieved))
|
||||||
|
for _, p := range retrieved {
|
||||||
|
stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
|
||||||
|
}
|
||||||
|
|
||||||
|
var unchanged, contentChanged, unknownIDs []Chunk
|
||||||
|
for pid, c := range incoming {
|
||||||
|
storedHash, found := stored[pid]
|
||||||
|
switch {
|
||||||
|
case found && storedHash == c.ContentHash:
|
||||||
|
unchanged = append(unchanged, c)
|
||||||
|
case found:
|
||||||
|
contentChanged = append(contentChanged, c)
|
||||||
|
default:
|
||||||
|
unknownIDs = append(unknownIDs, c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return incoming, unchanged, contentChanged, unknownIDs
|
||||||
|
}
|
||||||
|
|
||||||
|
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
|
||||||
|
// @block-end split-by-state
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
_, _, _, _ = incomingIDs, unchanged, contentChanged, unknownIDs
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start re-embed-changed
|
||||||
|
reEmbedChanged := func(contentChanged []Chunk) {
|
||||||
|
if len(contentChanged) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
points := make([]*qdrant.PointStruct, 0, len(contentChanged))
|
||||||
|
for _, c := range contentChanged {
|
||||||
|
points = append(points, &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// @block-end re-embed-changed
|
||||||
|
|
||||||
|
// @block-start reuse-or-add
|
||||||
|
// reuse an existing embedding when the same text is already stored; embed only what is new
|
||||||
|
reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
|
||||||
|
reused, added := 0, 0
|
||||||
|
|
||||||
|
for _, c := range unknownIDs {
|
||||||
|
sameText := &qdrant.Filter{
|
||||||
|
Must: []*qdrant.Condition{
|
||||||
|
qdrant.NewMatch("content_hash", c.ContentHash),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Filter: sameText,
|
||||||
|
Limit: qdrant.PtrOf(uint32(1)),
|
||||||
|
WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
|
||||||
|
WithVectors: qdrant.NewWithVectors(true),
|
||||||
|
})
|
||||||
|
if err != nil { panic(err) } // @hide
|
||||||
|
|
||||||
|
var point *qdrant.PointStruct
|
||||||
|
if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
|
||||||
|
}
|
||||||
|
reused++
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = &qdrant.PointStruct{
|
||||||
|
Id: qdrant.NewID(c.PointID),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
|
||||||
|
Payload: qdrant.NewValueMap(payload(c, "")),
|
||||||
|
}
|
||||||
|
added++
|
||||||
|
}
|
||||||
|
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{point},
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
return reused, added
|
||||||
|
}
|
||||||
|
// @block-end reuse-or-add
|
||||||
|
|
||||||
|
// @block-start delete-gone
|
||||||
|
// remove every point the current crawl no longer contains, return how many
|
||||||
|
deleteGone := func(incomingIDs map[string]Chunk) int {
|
||||||
|
if len(incomingIDs) == 0 {
|
||||||
|
panic("Refusing to delete from an empty source snapshot.")
|
||||||
|
}
|
||||||
|
|
||||||
|
ids := make([]*qdrant.PointId, 0, len(incomingIDs))
|
||||||
|
for pid := range incomingIDs {
|
||||||
|
ids = append(ids, qdrant.NewID(pid))
|
||||||
|
}
|
||||||
|
stale := &qdrant.Filter{
|
||||||
|
MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
|
||||||
|
}
|
||||||
|
|
||||||
|
toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Filter: stale,
|
||||||
|
})
|
||||||
|
if err != nil { panic(err) } // @hide
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.Delete(context.Background(), &qdrant.DeletePoints{
|
||||||
|
CollectionName: COLLECTION,
|
||||||
|
Points: qdrant.NewPointsSelectorFilter(stale),
|
||||||
|
Wait: qdrant.PtrOf(true),
|
||||||
|
})
|
||||||
|
return int(toDelete)
|
||||||
|
}
|
||||||
|
// @block-end delete-gone
|
||||||
|
|
||||||
|
// @block-start sync
|
||||||
|
sync := func(latestChunks []Chunk) map[string]int {
|
||||||
|
checkGate() // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
chunks := prepareChunksForSync(latestChunks)
|
||||||
|
incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
|
||||||
|
|
||||||
|
reEmbedChanged(contentChanged)
|
||||||
|
reused, added := reuseOrAdd(unknownIDs)
|
||||||
|
deleted := deleteGone(incomingIDs)
|
||||||
|
|
||||||
|
return map[string]int{
|
||||||
|
"unchanged": len(unchanged),
|
||||||
|
"re-embedded": len(contentChanged),
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end sync
|
||||||
|
|
||||||
|
// @block-start run-sync
|
||||||
|
run := sync(LATEST_CHUNKS)
|
||||||
|
fmt.Println(run)
|
||||||
|
// @block-end run-sync
|
||||||
|
}
|
||||||
+415
@@ -0,0 +1,415 @@
|
|||||||
|
package com.example.snippets_amalgamation;
|
||||||
|
|
||||||
|
import static io.qdrant.client.ConditionFactory.hasId;
|
||||||
|
import static io.qdrant.client.ConditionFactory.matchKeyword;
|
||||||
|
import static io.qdrant.client.PointIdFactory.id;
|
||||||
|
import static io.qdrant.client.QueryFactory.nearest;
|
||||||
|
import static io.qdrant.client.ValueFactory.value;
|
||||||
|
import static io.qdrant.client.VectorFactory.vector;
|
||||||
|
import static io.qdrant.client.VectorsFactory.vectors;
|
||||||
|
|
||||||
|
import io.qdrant.client.QdrantClient;
|
||||||
|
import io.qdrant.client.QdrantGrpcClient;
|
||||||
|
import io.qdrant.client.VectorOutputHelper;
|
||||||
|
import io.qdrant.client.WithPayloadSelectorFactory;
|
||||||
|
import io.qdrant.client.WithVectorsSelectorFactory;
|
||||||
|
import io.qdrant.client.grpc.Collections.CreateCollection;
|
||||||
|
import io.qdrant.client.grpc.Collections.Distance;
|
||||||
|
import io.qdrant.client.grpc.Collections.PayloadSchemaType;
|
||||||
|
import io.qdrant.client.grpc.Collections.VectorParams;
|
||||||
|
import io.qdrant.client.grpc.Collections.VectorsConfig;
|
||||||
|
import io.qdrant.client.grpc.Common.Filter;
|
||||||
|
import io.qdrant.client.grpc.JsonWithInt.Value;
|
||||||
|
import io.qdrant.client.grpc.Points.Document;
|
||||||
|
import io.qdrant.client.grpc.Points.PointStruct;
|
||||||
|
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.ScrollPoints;
|
||||||
|
import java.math.BigInteger;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.security.MessageDigest;
|
||||||
|
import java.time.OffsetDateTime;
|
||||||
|
import java.time.ZoneOffset;
|
||||||
|
import java.time.temporal.ChronoUnit;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.UUID;
|
||||||
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
|
public class Snippet {
|
||||||
|
|
||||||
|
// @block-start client-connection
|
||||||
|
static final String QDRANT_URL = System.getenv("QDRANT_URL");
|
||||||
|
static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
|
||||||
|
|
||||||
|
static final QdrantClient client =
|
||||||
|
new QdrantClient(
|
||||||
|
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
|
||||||
|
.withApiKey(QDRANT_API_KEY)
|
||||||
|
.build());
|
||||||
|
// @block-end client-connection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// data and text normalization are not the lesson of this tutorial:
|
||||||
|
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||||
|
static class Chunk {
|
||||||
|
String url;
|
||||||
|
String anchor;
|
||||||
|
int chunkNum;
|
||||||
|
String text;
|
||||||
|
String sectionUrl; // derived in prepareChunksForSync
|
||||||
|
String contentHash; // derived in prepareChunksForSync
|
||||||
|
String pointId; // derived in prepareChunksForSync
|
||||||
|
|
||||||
|
Chunk(String url, String anchor, int chunkNum, String text) {
|
||||||
|
this.url = url;
|
||||||
|
this.anchor = anchor;
|
||||||
|
this.chunkNum = chunkNum;
|
||||||
|
this.text = text;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static final List<Chunk> CHUNKS = List.of(
|
||||||
|
new Chunk(
|
||||||
|
"https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
"prerequisites",
|
||||||
|
0,
|
||||||
|
"Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ..."),
|
||||||
|
new Chunk(
|
||||||
|
"https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
"step-3-enable-an-admin-api-key",
|
||||||
|
0,
|
||||||
|
"Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ..."));
|
||||||
|
|
||||||
|
static String normalize(String text) {
|
||||||
|
return text.replaceAll("\\s+", " ").strip();
|
||||||
|
}
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-collection
|
||||||
|
static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
static final String PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
static final String COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
static void createCollection() throws Exception {
|
||||||
|
client.createCollectionAsync(
|
||||||
|
CreateCollection.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setVectorsConfig(
|
||||||
|
VectorsConfig.newBuilder()
|
||||||
|
.setParams(
|
||||||
|
VectorParams.newBuilder()
|
||||||
|
.setSize(384) // all-MiniLM-L6-v2 output dimension
|
||||||
|
.setDistance(Distance.Cosine)
|
||||||
|
.build())
|
||||||
|
.build())
|
||||||
|
.putAllMetadata(
|
||||||
|
Map.of(
|
||||||
|
"embedding_model", value(MODEL),
|
||||||
|
"pipeline_version", value(PIPELINE)))
|
||||||
|
.build()).get();
|
||||||
|
}
|
||||||
|
// @block-end create-collection
|
||||||
|
|
||||||
|
// @block-start check-gate
|
||||||
|
static void checkGate() throws Exception {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
Map<String, Value> meta =
|
||||||
|
client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
|
||||||
|
|
||||||
|
Value model = meta.get("embedding_model");
|
||||||
|
Value pipeline = meta.get("pipeline_version");
|
||||||
|
if (model == null || !MODEL.equals(model.getStringValue())
|
||||||
|
|| pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
|
||||||
|
throw new RuntimeException(
|
||||||
|
"collection was built by " + meta + ": full re-embed into a fresh collection required");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end check-gate
|
||||||
|
|
||||||
|
// @block-start identity-and-fingerprint
|
||||||
|
static String contentHash(String text) throws Exception {
|
||||||
|
byte[] digest = MessageDigest.getInstance("SHA-256")
|
||||||
|
.digest(text.getBytes(StandardCharsets.UTF_8));
|
||||||
|
return String.format("%064x", new BigInteger(1, digest));
|
||||||
|
}
|
||||||
|
|
||||||
|
static String pointId(String url, String anchor, int num) {
|
||||||
|
// name-based UUID (version 3); the same address always yields the same ID
|
||||||
|
return UUID.nameUUIDFromBytes(
|
||||||
|
(url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
static List<Chunk> prepareChunksForSync(List<Chunk> chunks) throws Exception {
|
||||||
|
List<Chunk> out = new ArrayList<>();
|
||||||
|
for (Chunk c : chunks) {
|
||||||
|
String text = normalize(c.text);
|
||||||
|
Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
|
||||||
|
prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
|
||||||
|
prepared.contentHash = contentHash(text);
|
||||||
|
prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
|
||||||
|
out.add(prepared);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
// @block-end identity-and-fingerprint
|
||||||
|
|
||||||
|
// @block-start payload
|
||||||
|
static Map<String, Value> payload(Chunk chunk, String lastUpdated) {
|
||||||
|
Map<String, Value> p = new HashMap<>();
|
||||||
|
p.put("url", value(chunk.url));
|
||||||
|
p.put("anchor", value(chunk.anchor));
|
||||||
|
p.put("chunk_num", value(chunk.chunkNum));
|
||||||
|
p.put("section_url", value(chunk.sectionUrl));
|
||||||
|
p.put("text", value(chunk.text));
|
||||||
|
p.put("content_hash", value(chunk.contentHash));
|
||||||
|
p.put("last_updated", value(lastUpdated != null
|
||||||
|
? lastUpdated
|
||||||
|
: OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
|
||||||
|
return p;
|
||||||
|
}
|
||||||
|
// @block-end payload
|
||||||
|
|
||||||
|
// @block-start payload-indexes
|
||||||
|
static void createPayloadIndexes() throws Exception {
|
||||||
|
for (String field : List.of("content_hash", "url", "section_url")) {
|
||||||
|
client.createPayloadIndexAsync(
|
||||||
|
COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end payload-indexes
|
||||||
|
|
||||||
|
// @block-start populate
|
||||||
|
static void populate() throws Exception {
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (Chunk c : prepareChunksForSync(CHUNKS)) {
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
client.upsertAsync(COLLECTION, points).get();
|
||||||
|
}
|
||||||
|
// @block-end populate
|
||||||
|
|
||||||
|
// @block-start search
|
||||||
|
static final String QUERY =
|
||||||
|
"Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
static void search() throws Exception {
|
||||||
|
client.queryAsync(
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(QUERY)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(3)
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
|
||||||
|
.build()).get();
|
||||||
|
}
|
||||||
|
// @block-end search
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||||
|
static List<Chunk> LATEST_CHUNKS;
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start split-by-state
|
||||||
|
static class SyncState {
|
||||||
|
Map<String, Chunk> incoming = new LinkedHashMap<>();
|
||||||
|
List<Chunk> unchanged = new ArrayList<>();
|
||||||
|
List<Chunk> contentChanged = new ArrayList<>();
|
||||||
|
List<Chunk> unknownIds = new ArrayList<>();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
static SyncState splitByState(List<Chunk> latestChunks) throws Exception {
|
||||||
|
SyncState state = new SyncState();
|
||||||
|
for (Chunk c : latestChunks) {
|
||||||
|
state.incoming.put(c.pointId, c);
|
||||||
|
}
|
||||||
|
|
||||||
|
Map<String, String> stored = new HashMap<>();
|
||||||
|
var points = client.retrieveAsync(
|
||||||
|
COLLECTION,
|
||||||
|
state.incoming.keySet().stream()
|
||||||
|
.map(pid -> id(UUID.fromString(pid)))
|
||||||
|
.collect(Collectors.toList()),
|
||||||
|
WithPayloadSelectorFactory.include(List.of("content_hash")),
|
||||||
|
WithVectorsSelectorFactory.enable(false),
|
||||||
|
null).get();
|
||||||
|
for (var p : points) {
|
||||||
|
stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
|
||||||
|
}
|
||||||
|
|
||||||
|
for (Map.Entry<String, Chunk> e : state.incoming.entrySet()) {
|
||||||
|
String pid = e.getKey();
|
||||||
|
Chunk c = e.getValue();
|
||||||
|
if (c.contentHash.equals(stored.get(pid))) {
|
||||||
|
state.unchanged.add(c);
|
||||||
|
} else if (stored.containsKey(pid)) {
|
||||||
|
state.contentChanged.add(c);
|
||||||
|
} else {
|
||||||
|
state.unknownIds.add(c);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return state;
|
||||||
|
}
|
||||||
|
// @block-end split-by-state
|
||||||
|
|
||||||
|
// @block-start re-embed-changed
|
||||||
|
static void reEmbedChanged(List<Chunk> contentChanged) throws Exception {
|
||||||
|
if (contentChanged.isEmpty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (Chunk c : contentChanged) {
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
client.upsertAsync(COLLECTION, points).get();
|
||||||
|
}
|
||||||
|
// @block-end re-embed-changed
|
||||||
|
|
||||||
|
// @block-start reuse-or-add
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
static int[] reuseOrAdd(List<Chunk> unknownIds) throws Exception {
|
||||||
|
int reused = 0;
|
||||||
|
int added = 0;
|
||||||
|
|
||||||
|
for (Chunk c : unknownIds) {
|
||||||
|
Filter sameText = Filter.newBuilder()
|
||||||
|
.addMust(matchKeyword("content_hash", c.contentHash))
|
||||||
|
.build();
|
||||||
|
|
||||||
|
var hits = client.scrollAsync(
|
||||||
|
ScrollPoints.newBuilder()
|
||||||
|
.setCollectionName(COLLECTION)
|
||||||
|
.setFilter(sameText)
|
||||||
|
.setLimit(1)
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
|
||||||
|
.setWithVectors(WithVectorsSelectorFactory.enable(true))
|
||||||
|
.build()).get().getResultList();
|
||||||
|
|
||||||
|
PointStruct point;
|
||||||
|
if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(vectors(vector(
|
||||||
|
VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
|
||||||
|
.getDataList())))
|
||||||
|
.putAllPayload(
|
||||||
|
payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
|
||||||
|
.build();
|
||||||
|
reused++;
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = PointStruct.newBuilder()
|
||||||
|
.setId(id(UUID.fromString(c.pointId)))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(c.text)
|
||||||
|
.setModel(MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(payload(c, null))
|
||||||
|
.build();
|
||||||
|
added++;
|
||||||
|
}
|
||||||
|
|
||||||
|
client.upsertAsync(COLLECTION, List.of(point)).get();
|
||||||
|
}
|
||||||
|
|
||||||
|
return new int[] {reused, added};
|
||||||
|
}
|
||||||
|
// @block-end reuse-or-add
|
||||||
|
|
||||||
|
// @block-start delete-gone
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
static long deleteGone(Map<String, Chunk> incomingIds) throws Exception {
|
||||||
|
if (incomingIds.isEmpty()) {
|
||||||
|
throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
Filter stale = Filter.newBuilder()
|
||||||
|
.addMustNot(hasId(
|
||||||
|
incomingIds.keySet().stream()
|
||||||
|
.map(pid -> id(UUID.fromString(pid)))
|
||||||
|
.collect(Collectors.toList())))
|
||||||
|
.build();
|
||||||
|
|
||||||
|
long toDelete = client.countAsync(COLLECTION, stale, true).get();
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.deleteAsync(COLLECTION, stale).get();
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
// @block-end delete-gone
|
||||||
|
|
||||||
|
// @block-start sync
|
||||||
|
static Map<String, Long> sync(List<Chunk> latestChunks) throws Exception {
|
||||||
|
checkGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
List<Chunk> chunks = prepareChunksForSync(latestChunks);
|
||||||
|
SyncState state = splitByState(chunks);
|
||||||
|
|
||||||
|
reEmbedChanged(state.contentChanged);
|
||||||
|
int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
|
||||||
|
long deleted = deleteGone(state.incoming);
|
||||||
|
|
||||||
|
return Map.of(
|
||||||
|
"unchanged", (long) state.unchanged.size(),
|
||||||
|
"re-embedded", (long) state.contentChanged.size(),
|
||||||
|
"reused_embedding", (long) reusedAdded[0],
|
||||||
|
"added", (long) reusedAdded[1],
|
||||||
|
"deleted", deleted);
|
||||||
|
}
|
||||||
|
// @block-end sync
|
||||||
|
|
||||||
|
// @block-start run-sync
|
||||||
|
static void runSync() throws Exception {
|
||||||
|
Map<String, Long> run = sync(LATEST_CHUNKS);
|
||||||
|
System.out.println(run);
|
||||||
|
}
|
||||||
|
// @block-end run-sync
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
public static void run() throws Exception {
|
||||||
|
createCollection();
|
||||||
|
createPayloadIndexes();
|
||||||
|
populate();
|
||||||
|
search();
|
||||||
|
|
||||||
|
LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
|
||||||
|
SyncState state = splitByState(LATEST_CHUNKS);
|
||||||
|
|
||||||
|
runSync();
|
||||||
|
// @hide-end
|
||||||
|
}
|
||||||
|
}
|
||||||
+261
@@ -0,0 +1,261 @@
|
|||||||
|
# @block-start client-connection
|
||||||
|
import os
|
||||||
|
|
||||||
|
from qdrant_client import QdrantClient, models
|
||||||
|
|
||||||
|
QDRANT_URL = os.getenv("QDRANT_URL")
|
||||||
|
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
||||||
|
|
||||||
|
client = QdrantClient(
|
||||||
|
url=QDRANT_URL,
|
||||||
|
api_key=QDRANT_API_KEY,
|
||||||
|
cloud_inference=True
|
||||||
|
)
|
||||||
|
# @block-end client-connection
|
||||||
|
|
||||||
|
# @hide-start
|
||||||
|
# data and text normalization are not the lesson of this tutorial:
|
||||||
|
# the full CHUNKS list and normalize() live in the tutorial notebook
|
||||||
|
CHUNKS = [
|
||||||
|
{
|
||||||
|
"url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
"anchor": "prerequisites",
|
||||||
|
"chunk_num": 0,
|
||||||
|
"text": "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
"anchor": "step-3-enable-an-admin-api-key",
|
||||||
|
"chunk_num": 0,
|
||||||
|
"text": "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
import re
|
||||||
|
import unicodedata
|
||||||
|
|
||||||
|
def normalize(text):
|
||||||
|
text = unicodedata.normalize("NFKC", text)
|
||||||
|
text = text.translate(dict.fromkeys(map(ord, "")))
|
||||||
|
return re.sub(r"\s+", " ", text).strip()
|
||||||
|
# @hide-end
|
||||||
|
|
||||||
|
# @block-start create-collection
|
||||||
|
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
||||||
|
PIPELINE = "docs-prep-pipeline-v1"
|
||||||
|
COLLECTION = "docs-sync-tutorial"
|
||||||
|
|
||||||
|
client.create_collection(
|
||||||
|
COLLECTION,
|
||||||
|
vectors_config=models.VectorParams(
|
||||||
|
size=384, # all-MiniLM-L6-v2 output dimension
|
||||||
|
distance=models.Distance.COSINE,
|
||||||
|
),
|
||||||
|
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
||||||
|
)
|
||||||
|
# @block-end create-collection
|
||||||
|
|
||||||
|
# @block-start check-gate
|
||||||
|
def check_gate():
|
||||||
|
# compare this pipeline's constants against what the collection records about itself
|
||||||
|
meta = client.get_collection(COLLECTION).config.metadata or {}
|
||||||
|
|
||||||
|
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
||||||
|
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
||||||
|
# @block-end check-gate
|
||||||
|
|
||||||
|
# @block-start identity-and-fingerprint
|
||||||
|
import hashlib
|
||||||
|
import uuid
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
def content_hash(text):
|
||||||
|
return hashlib.sha256(text.encode()).hexdigest()
|
||||||
|
|
||||||
|
def point_id(url, anchor, num):
|
||||||
|
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||||
|
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
||||||
|
|
||||||
|
def prepare_chunks_for_sync(chunks):
|
||||||
|
"""Derive both values (and the section address) for every raw chunk."""
|
||||||
|
out = []
|
||||||
|
for c in chunks:
|
||||||
|
text = normalize(c["text"])
|
||||||
|
out.append({
|
||||||
|
**c,
|
||||||
|
"text": text,
|
||||||
|
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
||||||
|
"content_hash": content_hash(text),
|
||||||
|
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
# @block-end identity-and-fingerprint
|
||||||
|
|
||||||
|
# @block-start payload
|
||||||
|
def payload(chunk, last_updated=None):
|
||||||
|
return {
|
||||||
|
"url": chunk["url"],
|
||||||
|
"anchor": chunk["anchor"],
|
||||||
|
"chunk_num": chunk["chunk_num"],
|
||||||
|
"section_url": chunk["section_url"],
|
||||||
|
"text": chunk["text"],
|
||||||
|
"content_hash": chunk["content_hash"],
|
||||||
|
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||||
|
}
|
||||||
|
# @block-end payload
|
||||||
|
|
||||||
|
# @block-start payload-indexes
|
||||||
|
for field in ("content_hash", "url", "section_url"):
|
||||||
|
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
||||||
|
# @block-end payload-indexes
|
||||||
|
|
||||||
|
# @block-start populate
|
||||||
|
client.upsert(COLLECTION, points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
for c in prepare_chunks_for_sync(CHUNKS)
|
||||||
|
], wait=True)
|
||||||
|
# @block-end populate
|
||||||
|
|
||||||
|
# @block-start search
|
||||||
|
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
||||||
|
|
||||||
|
client.query_points(
|
||||||
|
COLLECTION,
|
||||||
|
query=models.Document(text=QUERY, model=MODEL),
|
||||||
|
limit=3,
|
||||||
|
with_payload=["section_url", "text"],
|
||||||
|
)
|
||||||
|
# @block-end search
|
||||||
|
|
||||||
|
# @hide-start
|
||||||
|
# the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||||
|
LATEST_CHUNKS = prepare_chunks_for_sync(CHUNKS)
|
||||||
|
# @hide-end
|
||||||
|
|
||||||
|
# @block-start split-by-state
|
||||||
|
def split_by_state(latest_chunks):
|
||||||
|
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
||||||
|
incoming = {c["point_id"]: c for c in latest_chunks}
|
||||||
|
|
||||||
|
stored = {}
|
||||||
|
points = client.retrieve(
|
||||||
|
COLLECTION,
|
||||||
|
ids=list(incoming),
|
||||||
|
with_payload=["content_hash"],
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
for p in points:
|
||||||
|
stored[str(p.id)] = p.payload["content_hash"]
|
||||||
|
|
||||||
|
unchanged, content_changed, unknown_ids = [], [], []
|
||||||
|
for pid, c in incoming.items():
|
||||||
|
if stored.get(pid) == c["content_hash"]:
|
||||||
|
unchanged.append(c)
|
||||||
|
elif pid in stored:
|
||||||
|
content_changed.append(c)
|
||||||
|
else:
|
||||||
|
unknown_ids.append(c)
|
||||||
|
|
||||||
|
return incoming, unchanged, content_changed, unknown_ids
|
||||||
|
|
||||||
|
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
||||||
|
# @block-end split-by-state
|
||||||
|
|
||||||
|
# @block-start re-embed-changed
|
||||||
|
def re_embed_changed(content_changed):
|
||||||
|
if not content_changed:
|
||||||
|
return
|
||||||
|
client.upsert(COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
for c in content_changed],
|
||||||
|
wait=True)
|
||||||
|
# @block-end re-embed-changed
|
||||||
|
|
||||||
|
# @block-start reuse-or-add
|
||||||
|
def reuse_or_add(unknown_ids):
|
||||||
|
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
||||||
|
reused, added = 0, 0
|
||||||
|
|
||||||
|
for c in unknown_ids:
|
||||||
|
same_text = models.Filter(must=[
|
||||||
|
models.FieldCondition(
|
||||||
|
key="content_hash",
|
||||||
|
match=models.MatchValue(value=c["content_hash"]),
|
||||||
|
)
|
||||||
|
])
|
||||||
|
hits, _ = client.scroll(
|
||||||
|
COLLECTION,
|
||||||
|
scroll_filter=same_text,
|
||||||
|
limit=1,
|
||||||
|
with_payload=["last_updated"],
|
||||||
|
with_vectors=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if hits: # same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=hits[0].vector,
|
||||||
|
payload=payload(c, hits[0].payload["last_updated"]),
|
||||||
|
)
|
||||||
|
reused += 1
|
||||||
|
else: # genuinely new content: embed and insert
|
||||||
|
point = models.PointStruct(
|
||||||
|
id=c["point_id"],
|
||||||
|
vector=models.Document(text=c["text"], model=MODEL),
|
||||||
|
payload=payload(c),
|
||||||
|
)
|
||||||
|
added += 1
|
||||||
|
|
||||||
|
client.upsert(COLLECTION, points=[point], wait=True)
|
||||||
|
|
||||||
|
return reused, added
|
||||||
|
# @block-end reuse-or-add
|
||||||
|
|
||||||
|
# @block-start delete-gone
|
||||||
|
def delete_gone(incoming_ids):
|
||||||
|
"""Remove every point the current crawl no longer contains. Returns how many."""
|
||||||
|
if not incoming_ids:
|
||||||
|
raise ValueError("Refusing to delete from an empty source snapshot.")
|
||||||
|
|
||||||
|
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
||||||
|
|
||||||
|
to_delete = client.count(COLLECTION, count_filter=stale).count
|
||||||
|
|
||||||
|
# potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
||||||
|
return to_delete
|
||||||
|
# @block-end delete-gone
|
||||||
|
|
||||||
|
# @block-start sync
|
||||||
|
def sync(latest_chunks):
|
||||||
|
check_gate() # refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
chunks = prepare_chunks_for_sync(latest_chunks)
|
||||||
|
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
||||||
|
|
||||||
|
re_embed_changed(content_changed)
|
||||||
|
reused, added = reuse_or_add(unknown_ids)
|
||||||
|
deleted = delete_gone(incoming_ids)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"unchanged": len(unchanged),
|
||||||
|
"re-embedded": len(content_changed),
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
}
|
||||||
|
# @block-end sync
|
||||||
|
|
||||||
|
# @block-start run-sync
|
||||||
|
run = sync(LATEST_CHUNKS)
|
||||||
|
print(run)
|
||||||
|
# @block-end run-sync
|
||||||
+397
@@ -0,0 +1,397 @@
|
|||||||
|
use serde_json::{json, Value};
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use qdrant_client::qdrant::{
|
||||||
|
point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
|
||||||
|
CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
|
||||||
|
Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
|
||||||
|
Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
|
||||||
|
};
|
||||||
|
use qdrant_client::{Payload, Qdrant};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
pub async fn main() -> anyhow::Result<()> {
|
||||||
|
// @block-start client-connection
|
||||||
|
let qdrant_url = std::env::var("QDRANT_URL")?;
|
||||||
|
let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
|
||||||
|
|
||||||
|
let client = Qdrant::from_url(&qdrant_url)
|
||||||
|
.api_key(qdrant_api_key)
|
||||||
|
.build()?;
|
||||||
|
// @block-end client-connection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// data and text normalization are not the lesson of this tutorial:
|
||||||
|
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||||
|
#[derive(Clone, Default)]
|
||||||
|
struct Chunk {
|
||||||
|
url: String,
|
||||||
|
anchor: String,
|
||||||
|
chunk_num: u32,
|
||||||
|
text: String,
|
||||||
|
section_url: String,
|
||||||
|
content_hash: String,
|
||||||
|
point_id: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
let chunks: Vec<Chunk> = vec![
|
||||||
|
Chunk {
|
||||||
|
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
|
||||||
|
anchor: "prerequisites".into(),
|
||||||
|
chunk_num: 0,
|
||||||
|
text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...".into(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
Chunk {
|
||||||
|
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
|
||||||
|
anchor: "step-3-enable-an-admin-api-key".into(),
|
||||||
|
chunk_num: 0,
|
||||||
|
text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...".into(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
fn normalize(text: &str) -> String {
|
||||||
|
text.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||||
|
}
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-collection
|
||||||
|
const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
const PIPELINE: &str = "docs-prep-pipeline-v1";
|
||||||
|
const COLLECTION: &str = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
let mut metadata: HashMap<String, Value> = HashMap::new();
|
||||||
|
metadata.insert("embedding_model".to_string(), json!(MODEL));
|
||||||
|
metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
|
||||||
|
|
||||||
|
client
|
||||||
|
.create_collection(
|
||||||
|
CreateCollectionBuilder::new(COLLECTION)
|
||||||
|
.vectors_config(VectorParamsBuilder::new(
|
||||||
|
384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
Distance::Cosine,
|
||||||
|
))
|
||||||
|
.metadata(metadata),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
// @block-end create-collection
|
||||||
|
|
||||||
|
// @block-start check-gate
|
||||||
|
async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
let meta = client
|
||||||
|
.collection_info(COLLECTION)
|
||||||
|
.await?
|
||||||
|
.result
|
||||||
|
.and_then(|info| info.config)
|
||||||
|
.map(|config| config.metadata)
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
|
||||||
|
|| meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
|
||||||
|
!= Some(PIPELINE)
|
||||||
|
{
|
||||||
|
anyhow::bail!(
|
||||||
|
"collection was built by {meta:?}: full re-embed into a fresh collection required"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
// @block-end check-gate
|
||||||
|
|
||||||
|
// @block-start identity-and-fingerprint
|
||||||
|
fn content_hash(text: &str) -> String {
|
||||||
|
Sha256::digest(text.as_bytes())
|
||||||
|
.iter()
|
||||||
|
.map(|byte| format!("{byte:02x}"))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn point_id(url: &str, anchor: &str, num: u32) -> String {
|
||||||
|
// NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
||||||
|
uuid::Uuid::new_v5(
|
||||||
|
&uuid::Uuid::NAMESPACE_URL,
|
||||||
|
format!("{url}#{anchor}::{num}").as_bytes(),
|
||||||
|
)
|
||||||
|
.to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Derive both values (and the section address) for every raw chunk.
|
||||||
|
fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec<Chunk> {
|
||||||
|
chunks
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
let text = normalize(&c.text);
|
||||||
|
Chunk {
|
||||||
|
text: text.clone(),
|
||||||
|
section_url: if c.anchor.is_empty() {
|
||||||
|
c.url.clone()
|
||||||
|
} else {
|
||||||
|
format!("{}#{}", c.url, c.anchor)
|
||||||
|
},
|
||||||
|
content_hash: content_hash(&text),
|
||||||
|
point_id: point_id(&c.url, &c.anchor, c.chunk_num),
|
||||||
|
..c.clone()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
// @block-end identity-and-fingerprint
|
||||||
|
|
||||||
|
// @block-start payload
|
||||||
|
fn payload(chunk: &Chunk, last_updated: Option<String>) -> anyhow::Result<Payload> {
|
||||||
|
let last_updated = last_updated.unwrap_or_else(|| {
|
||||||
|
chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
|
||||||
|
});
|
||||||
|
Ok(Payload::try_from(serde_json::json!({
|
||||||
|
"url": chunk.url,
|
||||||
|
"anchor": chunk.anchor,
|
||||||
|
"chunk_num": chunk.chunk_num,
|
||||||
|
"section_url": chunk.section_url,
|
||||||
|
"text": chunk.text,
|
||||||
|
"content_hash": chunk.content_hash,
|
||||||
|
"last_updated": last_updated,
|
||||||
|
}))?)
|
||||||
|
}
|
||||||
|
// @block-end payload
|
||||||
|
|
||||||
|
// @block-start payload-indexes
|
||||||
|
for field in ["content_hash", "url", "section_url"] {
|
||||||
|
client
|
||||||
|
.create_field_index(CreateFieldIndexCollectionBuilder::new(
|
||||||
|
COLLECTION,
|
||||||
|
field,
|
||||||
|
FieldType::Keyword,
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
// @block-end payload-indexes
|
||||||
|
|
||||||
|
// @block-start populate
|
||||||
|
let points: Vec<PointStruct> = prepare_chunks_for_sync(&chunks)
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
Ok(PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
))
|
||||||
|
})
|
||||||
|
.collect::<anyhow::Result<_>>()?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||||
|
.await?;
|
||||||
|
// @block-end populate
|
||||||
|
|
||||||
|
// @block-start search
|
||||||
|
const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(COLLECTION)
|
||||||
|
.query(Query::new_nearest(Document::new(QUERY, MODEL)))
|
||||||
|
.limit(3)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec![
|
||||||
|
"section_url".to_string(),
|
||||||
|
"text".to_string(),
|
||||||
|
])),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
// @block-end search
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||||
|
let latest_chunks = prepare_chunks_for_sync(&chunks);
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start split-by-state
|
||||||
|
/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async fn split_by_state(
|
||||||
|
client: &Qdrant,
|
||||||
|
latest_chunks: &[Chunk],
|
||||||
|
) -> anyhow::Result<(HashMap<String, Chunk>, Vec<Chunk>, Vec<Chunk>, Vec<Chunk>)> {
|
||||||
|
let incoming: HashMap<String, Chunk> = latest_chunks
|
||||||
|
.iter()
|
||||||
|
.map(|c| (c.point_id.clone(), c.clone()))
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let ids: Vec<PointId> = incoming.keys().map(|id| id.as_str().into()).collect();
|
||||||
|
let points = client
|
||||||
|
.get_points(
|
||||||
|
GetPointsBuilder::new(COLLECTION, ids)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
|
||||||
|
.with_vectors(false),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let mut stored: HashMap<String, String> = HashMap::new();
|
||||||
|
for p in points.result {
|
||||||
|
let hash = p.get("content_hash").as_str().cloned();
|
||||||
|
if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
|
||||||
|
(p.id.and_then(|i| i.point_id_options), hash)
|
||||||
|
{
|
||||||
|
stored.insert(id, hash);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let (mut unchanged, mut content_changed, mut unknown_ids) =
|
||||||
|
(Vec::new(), Vec::new(), Vec::new());
|
||||||
|
for (pid, c) in &incoming {
|
||||||
|
if stored.get(pid) == Some(&c.content_hash) {
|
||||||
|
unchanged.push(c.clone());
|
||||||
|
} else if stored.contains_key(pid) {
|
||||||
|
content_changed.push(c.clone());
|
||||||
|
} else {
|
||||||
|
unknown_ids.push(c.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((incoming, unchanged, content_changed, unknown_ids))
|
||||||
|
}
|
||||||
|
|
||||||
|
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||||
|
split_by_state(&client, &latest_chunks).await?;
|
||||||
|
// @block-end split-by-state
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
_ = (&incoming_ids, &unchanged, &content_changed, &unknown_ids);
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start re-embed-changed
|
||||||
|
async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
|
||||||
|
if content_changed.is_empty() {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let points: Vec<PointStruct> = content_changed
|
||||||
|
.iter()
|
||||||
|
.map(|c| {
|
||||||
|
Ok(PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
))
|
||||||
|
})
|
||||||
|
.collect::<anyhow::Result<_>>()?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
// @block-end re-embed-changed
|
||||||
|
|
||||||
|
// @block-start reuse-or-add
|
||||||
|
/// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
|
||||||
|
let (mut reused, mut added) = (0, 0);
|
||||||
|
|
||||||
|
for c in unknown_ids {
|
||||||
|
let same_text =
|
||||||
|
Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
|
||||||
|
let hits = client
|
||||||
|
.scroll(
|
||||||
|
ScrollPointsBuilder::new(COLLECTION)
|
||||||
|
.filter(same_text)
|
||||||
|
.limit(1)
|
||||||
|
.with_payload(PayloadIncludeSelector::new(vec![
|
||||||
|
"last_updated".to_string()
|
||||||
|
]))
|
||||||
|
.with_vectors(true),
|
||||||
|
)
|
||||||
|
.await?
|
||||||
|
.result;
|
||||||
|
|
||||||
|
let point = if let Some(hit) = hits.into_iter().next() {
|
||||||
|
// same text, new address: copy the vector, keep its last_updated
|
||||||
|
let last_updated = hit.get("last_updated").as_str().cloned();
|
||||||
|
let vector: Vec<f32> = match hit.vectors.and_then(|v| v.vectors_options) {
|
||||||
|
Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
|
||||||
|
Some(vector_output::Vector::Dense(dense)) => dense.data,
|
||||||
|
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||||
|
},
|
||||||
|
_ => anyhow::bail!("expected a dense vector on the stored point"),
|
||||||
|
};
|
||||||
|
reused += 1;
|
||||||
|
PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
|
||||||
|
} else {
|
||||||
|
// genuinely new content: embed and insert
|
||||||
|
added += 1;
|
||||||
|
PointStruct::new(
|
||||||
|
c.point_id.clone(),
|
||||||
|
Document::new(&c.text, MODEL),
|
||||||
|
payload(c, None)?,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((reused, added))
|
||||||
|
}
|
||||||
|
// @block-end reuse-or-add
|
||||||
|
|
||||||
|
// @block-start delete-gone
|
||||||
|
/// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async fn delete_gone(
|
||||||
|
client: &Qdrant,
|
||||||
|
incoming_ids: &HashMap<String, Chunk>,
|
||||||
|
) -> anyhow::Result<u64> {
|
||||||
|
if incoming_ids.is_empty() {
|
||||||
|
anyhow::bail!("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
let stale = Filter::must_not([Condition::has_id(
|
||||||
|
incoming_ids.keys().map(|id| PointId::from(id.as_str())),
|
||||||
|
)]);
|
||||||
|
|
||||||
|
let to_delete = client
|
||||||
|
.count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
|
||||||
|
.await?
|
||||||
|
.result
|
||||||
|
.map(|r| r.count)
|
||||||
|
.unwrap_or(0);
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
client
|
||||||
|
.delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
|
||||||
|
.await?;
|
||||||
|
Ok(to_delete)
|
||||||
|
}
|
||||||
|
// @block-end delete-gone
|
||||||
|
|
||||||
|
// @block-start sync
|
||||||
|
async fn sync(
|
||||||
|
client: &Qdrant,
|
||||||
|
latest_chunks: &[Chunk],
|
||||||
|
) -> anyhow::Result<HashMap<&'static str, usize>> {
|
||||||
|
check_gate(client).await?; // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
let chunks = prepare_chunks_for_sync(latest_chunks);
|
||||||
|
let (incoming_ids, unchanged, content_changed, unknown_ids) =
|
||||||
|
split_by_state(client, &chunks).await?;
|
||||||
|
|
||||||
|
re_embed_changed(client, &content_changed).await?;
|
||||||
|
let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
|
||||||
|
let deleted = delete_gone(client, &incoming_ids).await?;
|
||||||
|
|
||||||
|
Ok(HashMap::from([
|
||||||
|
("unchanged", unchanged.len()),
|
||||||
|
("re-embedded", content_changed.len()),
|
||||||
|
("reused_embedding", reused),
|
||||||
|
("added", added),
|
||||||
|
("deleted", deleted as usize),
|
||||||
|
]))
|
||||||
|
}
|
||||||
|
// @block-end sync
|
||||||
|
|
||||||
|
// @block-start run-sync
|
||||||
|
let run = sync(&client, &latest_chunks).await?;
|
||||||
|
println!("{run:?}");
|
||||||
|
// @block-end run-sync
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
+288
@@ -0,0 +1,288 @@
|
|||||||
|
// @block-start client-connection
|
||||||
|
import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
|
||||||
|
|
||||||
|
const QDRANT_URL = process.env.QDRANT_URL;
|
||||||
|
const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
|
||||||
|
|
||||||
|
const client = new QdrantClient({
|
||||||
|
url: QDRANT_URL,
|
||||||
|
apiKey: QDRANT_API_KEY,
|
||||||
|
});
|
||||||
|
// @block-end client-connection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// data and text normalization are not the lesson of this tutorial:
|
||||||
|
// the full CHUNKS list and normalize() live in the tutorial notebook
|
||||||
|
const CHUNKS = [
|
||||||
|
{
|
||||||
|
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
anchor: "prerequisites",
|
||||||
|
chunk_num: 0,
|
||||||
|
text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
|
||||||
|
anchor: "step-3-enable-an-admin-api-key",
|
||||||
|
chunk_num: 0,
|
||||||
|
text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
function normalize(text: string): string {
|
||||||
|
return text
|
||||||
|
.normalize("NFKC")
|
||||||
|
.replace(/[\u200B\u200C\u200D\uFEFF\u00AD]/g, "")
|
||||||
|
.replace(/\s+/g, " ")
|
||||||
|
.trim();
|
||||||
|
}
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-collection
|
||||||
|
const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
|
||||||
|
const PIPELINE = "docs-prep-pipeline-v1";
|
||||||
|
const COLLECTION = "docs-sync-tutorial";
|
||||||
|
|
||||||
|
await client.createCollection(COLLECTION, {
|
||||||
|
vectors: {
|
||||||
|
size: 384, // all-MiniLM-L6-v2 output dimension
|
||||||
|
distance: "Cosine",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
await client.updateCollection(COLLECTION, {
|
||||||
|
metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
|
||||||
|
});
|
||||||
|
// @block-end create-collection
|
||||||
|
|
||||||
|
// @block-start check-gate
|
||||||
|
async function checkGate() {
|
||||||
|
// compare this pipeline's constants against what the collection records about itself
|
||||||
|
const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
|
||||||
|
{}) as Record<string, unknown>;
|
||||||
|
|
||||||
|
if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
|
||||||
|
throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end check-gate
|
||||||
|
|
||||||
|
// @block-start identity-and-fingerprint
|
||||||
|
import { createHash } from "node:crypto";
|
||||||
|
|
||||||
|
type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
|
||||||
|
type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
|
||||||
|
|
||||||
|
function contentHash(text: string): string {
|
||||||
|
return createHash("sha256").update(text).digest("hex");
|
||||||
|
}
|
||||||
|
|
||||||
|
// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
|
||||||
|
function pointId(url: string, anchor: string, num: number): string {
|
||||||
|
// Qdrant accepts any well-formed UUID as a point ID:
|
||||||
|
// hash the address, format the digest as a UUID, and the same address always yields the same ID
|
||||||
|
const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
|
||||||
|
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Derive both values (and the section address) for every raw chunk.
|
||||||
|
function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
|
||||||
|
return chunks.map((c) => {
|
||||||
|
const text = normalize(c.text);
|
||||||
|
return {
|
||||||
|
...c,
|
||||||
|
text,
|
||||||
|
section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
|
||||||
|
content_hash: contentHash(text),
|
||||||
|
point_id: pointId(c.url, c.anchor, c.chunk_num),
|
||||||
|
};
|
||||||
|
});
|
||||||
|
}
|
||||||
|
// @block-end identity-and-fingerprint
|
||||||
|
|
||||||
|
// @block-start payload
|
||||||
|
function payload(chunk: SyncChunk, lastUpdated?: string) {
|
||||||
|
return {
|
||||||
|
url: chunk.url,
|
||||||
|
anchor: chunk.anchor,
|
||||||
|
chunk_num: chunk.chunk_num,
|
||||||
|
section_url: chunk.section_url,
|
||||||
|
text: chunk.text,
|
||||||
|
content_hash: chunk.content_hash,
|
||||||
|
last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
// @block-end payload
|
||||||
|
|
||||||
|
// @block-start payload-indexes
|
||||||
|
for (const field of ["content_hash", "url", "section_url"]) {
|
||||||
|
await client.createPayloadIndex(COLLECTION, {
|
||||||
|
field_name: field,
|
||||||
|
field_schema: "keyword",
|
||||||
|
});
|
||||||
|
}
|
||||||
|
// @block-end payload-indexes
|
||||||
|
|
||||||
|
// @block-start populate
|
||||||
|
await client.upsert(COLLECTION, {
|
||||||
|
points: prepareChunksForSync(CHUNKS).map((c) => ({
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
})),
|
||||||
|
wait: true,
|
||||||
|
});
|
||||||
|
// @block-end populate
|
||||||
|
|
||||||
|
// @block-start search
|
||||||
|
const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
|
||||||
|
|
||||||
|
await client.query(COLLECTION, {
|
||||||
|
query: { text: QUERY, model: MODEL },
|
||||||
|
limit: 3,
|
||||||
|
with_payload: ["section_url", "text"],
|
||||||
|
});
|
||||||
|
// @block-end search
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
|
||||||
|
const LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start split-by-state
|
||||||
|
// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
|
||||||
|
async function splitByState(latestChunks: SyncChunk[]) {
|
||||||
|
const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
|
||||||
|
|
||||||
|
const stored = new Map<string, string>();
|
||||||
|
const points = await client.retrieve(COLLECTION, {
|
||||||
|
ids: [...incoming.keys()],
|
||||||
|
with_payload: ["content_hash"],
|
||||||
|
with_vector: false,
|
||||||
|
});
|
||||||
|
for (const p of points) {
|
||||||
|
stored.set(String(p.id), p.payload?.content_hash as string);
|
||||||
|
}
|
||||||
|
|
||||||
|
const unchanged: SyncChunk[] = [];
|
||||||
|
const contentChanged: SyncChunk[] = [];
|
||||||
|
const unknownIds: SyncChunk[] = [];
|
||||||
|
for (const [pid, c] of incoming) {
|
||||||
|
if (stored.get(pid) === c.content_hash) {
|
||||||
|
unchanged.push(c);
|
||||||
|
} else if (stored.has(pid)) {
|
||||||
|
contentChanged.push(c);
|
||||||
|
} else {
|
||||||
|
unknownIds.push(c);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return { incoming, unchanged, contentChanged, unknownIds };
|
||||||
|
}
|
||||||
|
|
||||||
|
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
|
||||||
|
// @block-end split-by-state
|
||||||
|
|
||||||
|
// @block-start re-embed-changed
|
||||||
|
async function reEmbedChanged(contentChanged: SyncChunk[]) {
|
||||||
|
if (contentChanged.length === 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
await client.upsert(COLLECTION, {
|
||||||
|
points: contentChanged.map((c) => ({
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
})),
|
||||||
|
wait: true,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
// @block-end re-embed-changed
|
||||||
|
|
||||||
|
// @block-start reuse-or-add
|
||||||
|
// Reuse an existing embedding when the same text is already stored; embed only what is new.
|
||||||
|
async function reuseOrAdd(unknownIds: SyncChunk[]) {
|
||||||
|
let reused = 0;
|
||||||
|
let added = 0;
|
||||||
|
|
||||||
|
for (const c of unknownIds) {
|
||||||
|
const sameText = {
|
||||||
|
must: [
|
||||||
|
{
|
||||||
|
key: "content_hash",
|
||||||
|
match: { value: c.content_hash },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
};
|
||||||
|
const hits = (await client.scroll(COLLECTION, {
|
||||||
|
filter: sameText,
|
||||||
|
limit: 1,
|
||||||
|
with_payload: ["last_updated"],
|
||||||
|
with_vector: true,
|
||||||
|
})).points;
|
||||||
|
|
||||||
|
let point: Schemas["PointStruct"];
|
||||||
|
if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
|
||||||
|
point = {
|
||||||
|
id: c.point_id,
|
||||||
|
vector: hits[0].vector as number[],
|
||||||
|
payload: payload(c, hits[0].payload?.last_updated as string),
|
||||||
|
};
|
||||||
|
reused += 1;
|
||||||
|
} else { // genuinely new content: embed and insert
|
||||||
|
point = {
|
||||||
|
id: c.point_id,
|
||||||
|
vector: { text: c.text, model: MODEL },
|
||||||
|
payload: payload(c),
|
||||||
|
};
|
||||||
|
added += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
await client.upsert(COLLECTION, { points: [point], wait: true });
|
||||||
|
}
|
||||||
|
|
||||||
|
return { reused, added };
|
||||||
|
}
|
||||||
|
// @block-end reuse-or-add
|
||||||
|
|
||||||
|
// @block-start delete-gone
|
||||||
|
// Remove every point the current crawl no longer contains. Returns how many.
|
||||||
|
async function deleteGone(incoming: Map<string, SyncChunk>) {
|
||||||
|
if (incoming.size === 0) {
|
||||||
|
throw new Error("Refusing to delete from an empty source snapshot.");
|
||||||
|
}
|
||||||
|
|
||||||
|
const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
|
||||||
|
|
||||||
|
const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
|
||||||
|
|
||||||
|
// potential check against a threshold to avoid accidental mass deletion could be added here
|
||||||
|
await client.delete(COLLECTION, { filter: stale, wait: true });
|
||||||
|
return toDelete;
|
||||||
|
}
|
||||||
|
// @block-end delete-gone
|
||||||
|
|
||||||
|
// @block-start sync
|
||||||
|
async function sync(latestChunks: RawChunk[]) {
|
||||||
|
await checkGate(); // refuse to mix embedding models or pipeline versions
|
||||||
|
|
||||||
|
const chunks = prepareChunksForSync(latestChunks);
|
||||||
|
const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
|
||||||
|
|
||||||
|
await reEmbedChanged(contentChanged);
|
||||||
|
const { reused, added } = await reuseOrAdd(unknownIds);
|
||||||
|
const deleted = await deleteGone(incoming);
|
||||||
|
|
||||||
|
return {
|
||||||
|
"unchanged": unchanged.length,
|
||||||
|
"re-embedded": contentChanged.length,
|
||||||
|
"reused_embedding": reused,
|
||||||
|
"added": added,
|
||||||
|
"deleted": deleted,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
// @block-end sync
|
||||||
|
|
||||||
|
// @block-start run-sync
|
||||||
|
const run = await sync(LATEST_CHUNKS);
|
||||||
|
console.log(run);
|
||||||
|
// @block-end run-sync
|
||||||
+15
-219
@@ -35,27 +35,13 @@ The tutorial has an accompanying [notebook](https://github.com/qdrant/examples/b
|
|||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
|
|
||||||
```python
|
Install the [Qdrant client of your choice](/documentation/interfaces/#client-libraries).
|
||||||
%pip install -q "qdrant-client>=1.18"
|
|
||||||
```
|
|
||||||
|
|
||||||
We use Qdrant Cloud and its [Free Embedding Inference](/documentation/cloud/inference/#free-embedding-models).
|
We use Qdrant Cloud and its [Free Embedding Inference](/documentation/cloud/inference/#free-embedding-models).
|
||||||
Create a Free Tier [Qdrant Cloud cluster](https://cloud.qdrant.io/) and set `QDRANT_URL` and `QDRANT_API_KEY` in your environment.
|
Create a Free Tier [Qdrant Cloud cluster](https://cloud.qdrant.io/) and set `QDRANT_URL` and `QDRANT_API_KEY` in your environment.
|
||||||
|
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="client-connection" >}}
|
||||||
import os
|
|
||||||
from qdrant_client import QdrantClient, models
|
|
||||||
|
|
||||||
QDRANT_URL = os.getenv("QDRANT_URL")
|
|
||||||
QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
|
|
||||||
|
|
||||||
client = QdrantClient(
|
|
||||||
url=QDRANT_URL,
|
|
||||||
api_key=QDRANT_API_KEY,
|
|
||||||
cloud_inference=True
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
## The Data: Qdrant Documentation
|
## The Data: Qdrant Documentation
|
||||||
|
|
||||||
@@ -133,31 +119,11 @@ Vectors produced by different embedding models, or by the same model over differ
|
|||||||
|
|
||||||
Let's consider a simple guardrail: save which model and which pipeline version produced the data points, in [**collection metadata**](/documentation/manage-data/collections/#collection-metadata), and verify against it. If one of the two changed, we need to trigger full collection re-embedding.
|
Let's consider a simple guardrail: save which model and which pipeline version produced the data points, in [**collection metadata**](/documentation/manage-data/collections/#collection-metadata), and verify against it. If one of the two changed, we need to trigger full collection re-embedding.
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="create-collection" >}}
|
||||||
MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
|
||||||
PIPELINE = "docs-prep-pipeline-v1"
|
|
||||||
COLLECTION = "docs-sync-tutorial"
|
|
||||||
|
|
||||||
client.create_collection(
|
|
||||||
COLLECTION,
|
|
||||||
vectors_config=models.VectorParams(
|
|
||||||
size=384, # all-MiniLM-L6-v2 output dimension
|
|
||||||
distance=models.Distance.COSINE,
|
|
||||||
),
|
|
||||||
metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
The gate against mixing embedding generations is then a simple check at the start of every run:
|
The gate against mixing embedding generations is then a simple check at the start of every run:
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="check-gate" >}}
|
||||||
def check_gate():
|
|
||||||
# compare this pipeline's constants against what the collection records about itself
|
|
||||||
meta = client.get_collection(COLLECTION).config.metadata or {}
|
|
||||||
|
|
||||||
if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
|
|
||||||
raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
|
|
||||||
```
|
|
||||||
|
|
||||||
## Characteristics of a Document Chunk
|
## Characteristics of a Document Chunk
|
||||||
|
|
||||||
@@ -174,32 +140,7 @@ Hence every record should get two derived values:
|
|||||||
- **Content fingerprint**, like SHA-256 of the text. It changes if a single character changes, and never otherwise. Comparing fingerprints answers "*Is it the same content?*" without comparing texts.
|
- **Content fingerprint**, like SHA-256 of the text. It changes if a single character changes, and never otherwise. Comparing fingerprints answers "*Is it the same content?*" without comparing texts.
|
||||||
- **Deterministic ID** for position in documentation. For example, `url + "#" + anchor + "::" + chunk_num` turned into a UUID, one of the two point ID formats Qdrant accepts. Comparing IDs answers "*Is this content still at the same position?*".
|
- **Deterministic ID** for position in documentation. For example, `url + "#" + anchor + "::" + chunk_num` turned into a UUID, one of the two point ID formats Qdrant accepts. Comparing IDs answers "*Is this content still at the same position?*".
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="identity-and-fingerprint" >}}
|
||||||
import hashlib
|
|
||||||
import uuid
|
|
||||||
from datetime import datetime, timezone
|
|
||||||
|
|
||||||
def content_hash(text):
|
|
||||||
return hashlib.sha256(text.encode()).hexdigest()
|
|
||||||
|
|
||||||
def point_id(url, anchor, num):
|
|
||||||
# NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
|
|
||||||
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
|
|
||||||
|
|
||||||
def prepare_chunks_for_sync(chunks):
|
|
||||||
"""Derive both values (and the section address) for every raw chunk."""
|
|
||||||
out = []
|
|
||||||
for c in chunks:
|
|
||||||
text = normalize(c["text"])
|
|
||||||
out.append({
|
|
||||||
**c,
|
|
||||||
"text": text,
|
|
||||||
"section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
|
|
||||||
"content_hash": content_hash(text),
|
|
||||||
"point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
|
|
||||||
})
|
|
||||||
return out
|
|
||||||
```
|
|
||||||
Example:
|
Example:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
@@ -218,56 +159,24 @@ Additionally, a point can be described by the following fields:
|
|||||||
<details>
|
<details>
|
||||||
<summary>payload() implementation</summary>
|
<summary>payload() implementation</summary>
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload" >}}
|
||||||
def payload(chunk, last_updated=None):
|
|
||||||
return {
|
|
||||||
"url": chunk["url"],
|
|
||||||
"anchor": chunk["anchor"],
|
|
||||||
"chunk_num": chunk["chunk_num"],
|
|
||||||
"section_url": chunk["section_url"],
|
|
||||||
"text": chunk["text"],
|
|
||||||
"content_hash": chunk["content_hash"],
|
|
||||||
"last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
</details>
|
</details>
|
||||||
|
|
||||||
For all the payload fields used for filtering or grouping we need to create a [**payload index**](/documentation/manage-data/indexing/).
|
For all the payload fields used for filtering or grouping we need to create a [**payload index**](/documentation/manage-data/indexing/).
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload-indexes" >}}
|
||||||
for field in ("content_hash", "url", "section_url"):
|
|
||||||
client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
|
|
||||||
```
|
|
||||||
|
|
||||||
## Populate Collection
|
## Populate Collection
|
||||||
|
|
||||||
Populate the collection with the whole documentation.
|
Populate the collection with the whole documentation.
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="populate" >}}
|
||||||
client.upsert(COLLECTION, points=[
|
|
||||||
models.PointStruct(
|
|
||||||
id=c["point_id"],
|
|
||||||
vector=models.Document(text=c["text"], model=MODEL), # Cloud Inference embeds text server-side
|
|
||||||
payload=payload(c),
|
|
||||||
)
|
|
||||||
for c in prepare_chunks_for_sync(CHUNKS)
|
|
||||||
], wait=True)
|
|
||||||
```
|
|
||||||
|
|
||||||
<details>
|
<details>
|
||||||
<summary>Test the search against it</summary>
|
<summary>Test the search against it</summary>
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="search" >}}
|
||||||
QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
|
|
||||||
|
|
||||||
client.query_points(
|
|
||||||
COLLECTION,
|
|
||||||
query=models.Document(text=QUERY, model=MODEL),
|
|
||||||
limit=3,
|
|
||||||
with_payload=["section_url", "text"],
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
You should get something like:
|
You should get something like:
|
||||||
|
|
||||||
@@ -354,35 +263,7 @@ We now check every incoming chunk against the collection: does its ID (address)
|
|||||||
|
|
||||||
[`retrieve`](/documentation/manage-data/points/) fetches points by ID. At corpus scale you would batch the IDs.
|
[`retrieve`](/documentation/manage-data/points/) fetches points by ID. At corpus scale you would batch the IDs.
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="split-by-state" >}}
|
||||||
def split_by_state(latest_chunks):
|
|
||||||
"""Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
|
|
||||||
incoming = {c["point_id"]: c for c in latest_chunks}
|
|
||||||
|
|
||||||
stored = {}
|
|
||||||
points = client.retrieve(
|
|
||||||
COLLECTION,
|
|
||||||
ids=list(incoming),
|
|
||||||
with_payload=["content_hash"],
|
|
||||||
with_vectors=False,
|
|
||||||
)
|
|
||||||
for p in points:
|
|
||||||
stored[str(p.id)] = p.payload["content_hash"]
|
|
||||||
|
|
||||||
unchanged, content_changed, unknown_ids = [], [], []
|
|
||||||
for pid, c in incoming.items():
|
|
||||||
if stored.get(pid) == c["content_hash"]:
|
|
||||||
unchanged.append(c)
|
|
||||||
elif pid in stored:
|
|
||||||
content_changed.append(c)
|
|
||||||
else:
|
|
||||||
unknown_ids.append(c)
|
|
||||||
|
|
||||||
return incoming, unchanged, content_changed, unknown_ids
|
|
||||||
|
|
||||||
|
|
||||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
|
|
||||||
```
|
|
||||||
|
|
||||||
### Case 1: Unchanged, Do Nothing
|
### Case 1: Unchanged, Do Nothing
|
||||||
|
|
||||||
@@ -393,20 +274,7 @@ These chunks carry the same fingerprint as before.
|
|||||||
The chunk about Step 3 exists under a known ID (it didn't change its position on the docs website) but carries new information.
|
The chunk about Step 3 exists under a known ID (it didn't change its position on the docs website) but carries new information.
|
||||||
Use `upsert`: writing a point under an existing ID replaces it.
|
Use `upsert`: writing a point under an existing ID replaces it.
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="re-embed-changed" >}}
|
||||||
def re_embed_changed(content_changed):
|
|
||||||
if not content_changed:
|
|
||||||
return
|
|
||||||
client.upsert(COLLECTION,
|
|
||||||
points=[
|
|
||||||
models.PointStruct(
|
|
||||||
id=c["point_id"],
|
|
||||||
vector=models.Document(text=c["text"], model=MODEL),
|
|
||||||
payload=payload(c),
|
|
||||||
)
|
|
||||||
for c in content_changed],
|
|
||||||
wait=True)
|
|
||||||
```
|
|
||||||
|
|
||||||
### Cases 3 and 4: ID Is Not Present in the Collection
|
### Cases 3 and 4: ID Is Not Present in the Collection
|
||||||
|
|
||||||
@@ -418,45 +286,7 @@ A filtered [`scroll`](/documentation/manage-data/points/) on `content_hash` answ
|
|||||||
|
|
||||||
**Note:** *This version performs one hash lookup per unknown chunk so the decision is easy to inspect. In production, batch hash lookups and point upserts.*
|
**Note:** *This version performs one hash lookup per unknown chunk so the decision is easy to inspect. In production, batch hash lookups and point upserts.*
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="reuse-or-add" >}}
|
||||||
def reuse_or_add(unknown_ids):
|
|
||||||
"""Reuse an existing embedding when the same text is already stored; embed only what is new."""
|
|
||||||
reused, added = 0, 0
|
|
||||||
|
|
||||||
for c in unknown_ids:
|
|
||||||
same_text = models.Filter(must=[
|
|
||||||
models.FieldCondition(
|
|
||||||
key="content_hash",
|
|
||||||
match=models.MatchValue(value=c["content_hash"]),
|
|
||||||
)
|
|
||||||
])
|
|
||||||
hits, _ = client.scroll(
|
|
||||||
COLLECTION,
|
|
||||||
scroll_filter=same_text,
|
|
||||||
limit=1,
|
|
||||||
with_payload=["last_updated"],
|
|
||||||
with_vectors=True,
|
|
||||||
)
|
|
||||||
|
|
||||||
if hits: # same text, new address: copy the vector, keep its last_updated
|
|
||||||
point = models.PointStruct(
|
|
||||||
id=c["point_id"],
|
|
||||||
vector=hits[0].vector,
|
|
||||||
payload=payload(c, hits[0].payload["last_updated"]),
|
|
||||||
)
|
|
||||||
reused += 1
|
|
||||||
else: # genuinely new content: embed and insert
|
|
||||||
point = models.PointStruct(
|
|
||||||
id=c["point_id"],
|
|
||||||
vector=models.Document(text=c["text"], model=MODEL),
|
|
||||||
payload=payload(c),
|
|
||||||
)
|
|
||||||
added += 1
|
|
||||||
|
|
||||||
client.upsert(COLLECTION, points=[point], wait=True)
|
|
||||||
|
|
||||||
return reused, added
|
|
||||||
```
|
|
||||||
|
|
||||||
What's important to notice: the old points, the migration page under its old URL, are still in the collection. They need to be removed, and that is the last case.
|
What's important to notice: the old points, the migration page under its old URL, are still in the collection. They need to be removed, and that is the last case.
|
||||||
|
|
||||||
@@ -471,51 +301,17 @@ Whatever LATEST_CHUNKS does not contain no longer exists at the source. The dele
|
|||||||
**Note:** Frequent re-embeddings and deletions don't degrade the index over time: background [optimizers](/documentation/ops-optimization/optimizer/) rebuild and merge index segments as changes accumulate.
|
**Note:** Frequent re-embeddings and deletions don't degrade the index over time: background [optimizers](/documentation/ops-optimization/optimizer/) rebuild and merge index segments as changes accumulate.
|
||||||
|
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="delete-gone" >}}
|
||||||
def delete_gone(incoming_ids):
|
|
||||||
"""Remove every point the current crawl no longer contains. Returns how many."""
|
|
||||||
if not incoming_ids:
|
|
||||||
raise ValueError("Refusing to delete from an empty source snapshot.")
|
|
||||||
|
|
||||||
stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
|
|
||||||
|
|
||||||
to_delete = client.count(COLLECTION, count_filter=stale).count
|
|
||||||
|
|
||||||
# potential check against a threshold to avoid accidental mass deletion could be added here
|
|
||||||
client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
|
|
||||||
return to_delete
|
|
||||||
```
|
|
||||||
|
|
||||||
## Run and Verify the Sync
|
## Run and Verify the Sync
|
||||||
|
|
||||||
The five cases, assembled from the functions defined above:
|
The five cases, assembled from the functions defined above:
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="sync" >}}
|
||||||
def sync(latest_chunks):
|
|
||||||
check_gate() # refuse to mix embedding models or pipeline versions
|
|
||||||
|
|
||||||
chunks = prepare_chunks_for_sync(latest_chunks)
|
|
||||||
incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
|
|
||||||
|
|
||||||
re_embed_changed(content_changed)
|
|
||||||
reused, added = reuse_or_add(unknown_ids)
|
|
||||||
deleted = delete_gone(incoming_ids)
|
|
||||||
|
|
||||||
return {
|
|
||||||
"unchanged": len(unchanged),
|
|
||||||
"re-embedded": len(content_changed),
|
|
||||||
"reused_embedding": reused,
|
|
||||||
"added": added,
|
|
||||||
"deleted": deleted,
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Run the sync.
|
Run the sync.
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="run-sync" >}}
|
||||||
run = sync(LATEST_CHUNKS)
|
|
||||||
print(run)
|
|
||||||
```
|
|
||||||
|
|
||||||
You should see something like:
|
You should see something like:
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user