mirror of
https://github.com/qdrant/landing_page.git
synced 2026-10-04 10:28:29 +02:00
Model migration tutorial: use insert-only update mode and Cloud Inference (#2168)
* Switch to insert-only mode instead of conditional upserts * Make code snippets testable; use Cloud Inference * Use regular upserts instead of batch_update_points * Add snippets for TS, Rust, Java, C#, and Go
This commit is contained in:
+154
@@ -0,0 +1,154 @@
|
|||||||
|
using Qdrant.Client;
|
||||||
|
using Qdrant.Client.Grpc;
|
||||||
|
|
||||||
|
public class Snippet
|
||||||
|
{
|
||||||
|
public static async Task Run()
|
||||||
|
{
|
||||||
|
// @hide-start
|
||||||
|
var client = new QdrantClient(
|
||||||
|
host: "",
|
||||||
|
port: 6334,
|
||||||
|
https: true,
|
||||||
|
apiKey: ""
|
||||||
|
);
|
||||||
|
|
||||||
|
string NEW_COLLECTION = "new_collection";
|
||||||
|
string OLD_COLLECTION = "old_collection";
|
||||||
|
|
||||||
|
string OLD_MODEL = "sentence-transformers/all-minilm-l6-v2";
|
||||||
|
string NEW_MODEL = "qdrant/clip-vit-b-32-text";
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-new-collection
|
||||||
|
await client.CreateCollectionAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
vectorsConfig: new VectorParams { Size = 512, Distance = Distance.Cosine }
|
||||||
|
);
|
||||||
|
// @block-end create-new-collection
|
||||||
|
|
||||||
|
// @block-start upsert-old-collection
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
points: new List<PointStruct>
|
||||||
|
{
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
Id = 1,
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = "Example document",
|
||||||
|
Model = OLD_MODEL
|
||||||
|
},
|
||||||
|
Payload = { ["text"] = "Example document" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
);
|
||||||
|
// @block-end upsert-old-collection
|
||||||
|
|
||||||
|
// @block-start upsert-new-collection
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
points: new List<PointStruct>
|
||||||
|
{
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
Id = 1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = "Example document",
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
Payload = { ["text"] = "Example document" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
);
|
||||||
|
// @block-end upsert-new-collection
|
||||||
|
|
||||||
|
// @block-start migrate-points
|
||||||
|
PointId? lastOffset = null;
|
||||||
|
uint limit = 100; // Number of points to read in each batch
|
||||||
|
bool reachedEnd = false;
|
||||||
|
|
||||||
|
while (!reachedEnd)
|
||||||
|
{
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
var scrollResult = await client.ScrollAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
limit: limit,
|
||||||
|
offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
payloadSelector: true,
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
vectorsSelector: false
|
||||||
|
);
|
||||||
|
|
||||||
|
var records = scrollResult.Result;
|
||||||
|
lastOffset = scrollResult.NextPageOffset;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
var points = new List<PointStruct>();
|
||||||
|
foreach (var record in records)
|
||||||
|
{
|
||||||
|
var text = record.Payload.ContainsKey("text")
|
||||||
|
? record.Payload["text"].StringValue
|
||||||
|
: "";
|
||||||
|
|
||||||
|
points.Add(new PointStruct
|
||||||
|
{
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
Id = record.Id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = text,
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
// Keep the original payload
|
||||||
|
Payload = { record.Payload }
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
await client.UpsertAsync(
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
CollectionName = NEW_COLLECTION,
|
||||||
|
Points = { points },
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpdateMode = UpdateMode.InsertOnly
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = (lastOffset == null);
|
||||||
|
}
|
||||||
|
// @block-end migrate-points
|
||||||
|
|
||||||
|
// @block-start search-old-collection
|
||||||
|
var results = await client.QueryAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
query: new Document
|
||||||
|
{
|
||||||
|
Text = "my query",
|
||||||
|
Model = OLD_MODEL
|
||||||
|
},
|
||||||
|
limit: 10
|
||||||
|
);
|
||||||
|
// @block-end search-old-collection
|
||||||
|
|
||||||
|
// @block-start search-new-collection
|
||||||
|
results = await client.QueryAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
query: new Document
|
||||||
|
{
|
||||||
|
Text = "my query",
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
limit: 10
|
||||||
|
);
|
||||||
|
// @block-end search-new-collection
|
||||||
|
}
|
||||||
|
}
|
||||||
+6
@@ -0,0 +1,6 @@
|
|||||||
|
```csharp
|
||||||
|
await client.CreateCollectionAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
vectorsConfig: new VectorParams { Size = 512, Distance = Distance.Cosine }
|
||||||
|
);
|
||||||
|
```
|
||||||
+9
@@ -0,0 +1,9 @@
|
|||||||
|
```go
|
||||||
|
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||||
|
Size: 512, // Size of the new embedding vectors
|
||||||
|
Distance: qdrant.Distance_Cosine,
|
||||||
|
}),
|
||||||
|
})
|
||||||
|
```
|
||||||
+7
@@ -0,0 +1,7 @@
|
|||||||
|
```java
|
||||||
|
client.createCollectionAsync(NEW_COLLECTION,
|
||||||
|
VectorParams.newBuilder()
|
||||||
|
.setSize(512) // Size of the new embedding vectors
|
||||||
|
.setDistance(Distance.Cosine) // Similarity function for the new model
|
||||||
|
.build()).get();
|
||||||
|
```
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
```python
|
||||||
|
client.create_collection(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
vectors_config=(
|
||||||
|
models.VectorParams(
|
||||||
|
size=512, # Size of the new embedding vectors
|
||||||
|
distance=models.Distance.COSINE # Similarity function for the new model
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
```
|
||||||
+8
@@ -0,0 +1,8 @@
|
|||||||
|
```rust
|
||||||
|
client
|
||||||
|
.create_collection(
|
||||||
|
CreateCollectionBuilder::new(new_collection)
|
||||||
|
.vectors_config(VectorParamsBuilder::new(512, Distance::Cosine)), // Size of the new embedding vectors
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+8
@@ -0,0 +1,8 @@
|
|||||||
|
```typescript
|
||||||
|
await client.createCollection(NEW_COLLECTION, {
|
||||||
|
vectors: {
|
||||||
|
size: 512, // Size of the new embedding vectors
|
||||||
|
distance: "Cosine", // Similarity function for the new model
|
||||||
|
},
|
||||||
|
});
|
||||||
|
```
|
||||||
+123
@@ -0,0 +1,123 @@
|
|||||||
|
```csharp
|
||||||
|
using Qdrant.Client;
|
||||||
|
using Qdrant.Client.Grpc;
|
||||||
|
|
||||||
|
await client.CreateCollectionAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
vectorsConfig: new VectorParams { Size = 512, Distance = Distance.Cosine }
|
||||||
|
);
|
||||||
|
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
points: new List<PointStruct>
|
||||||
|
{
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
Id = 1,
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = "Example document",
|
||||||
|
Model = OLD_MODEL
|
||||||
|
},
|
||||||
|
Payload = { ["text"] = "Example document" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
points: new List<PointStruct>
|
||||||
|
{
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
Id = 1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = "Example document",
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
Payload = { ["text"] = "Example document" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
PointId? lastOffset = null;
|
||||||
|
uint limit = 100; // Number of points to read in each batch
|
||||||
|
bool reachedEnd = false;
|
||||||
|
|
||||||
|
while (!reachedEnd)
|
||||||
|
{
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
var scrollResult = await client.ScrollAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
limit: limit,
|
||||||
|
offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
payloadSelector: true,
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
vectorsSelector: false
|
||||||
|
);
|
||||||
|
|
||||||
|
var records = scrollResult.Result;
|
||||||
|
lastOffset = scrollResult.NextPageOffset;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
var points = new List<PointStruct>();
|
||||||
|
foreach (var record in records)
|
||||||
|
{
|
||||||
|
var text = record.Payload.ContainsKey("text")
|
||||||
|
? record.Payload["text"].StringValue
|
||||||
|
: "";
|
||||||
|
|
||||||
|
points.Add(new PointStruct
|
||||||
|
{
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
Id = record.Id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = text,
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
// Keep the original payload
|
||||||
|
Payload = { record.Payload }
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
await client.UpsertAsync(
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
CollectionName = NEW_COLLECTION,
|
||||||
|
Points = { points },
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpdateMode = UpdateMode.InsertOnly
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = (lastOffset == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
var results = await client.QueryAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
query: new Document
|
||||||
|
{
|
||||||
|
Text = "my query",
|
||||||
|
Model = OLD_MODEL
|
||||||
|
},
|
||||||
|
limit: 10
|
||||||
|
);
|
||||||
|
|
||||||
|
results = await client.QueryAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
query: new Document
|
||||||
|
{
|
||||||
|
Text = "my query",
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
limit: 10
|
||||||
|
);
|
||||||
|
```
|
||||||
+114
@@ -0,0 +1,114 @@
|
|||||||
|
```go
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
|
||||||
|
"github.com/qdrant/go-client/qdrant"
|
||||||
|
)
|
||||||
|
|
||||||
|
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||||
|
Size: 512, // Size of the new embedding vectors
|
||||||
|
Distance: qdrant.Distance_Cosine,
|
||||||
|
}),
|
||||||
|
})
|
||||||
|
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{
|
||||||
|
{
|
||||||
|
Id: qdrant.NewIDNum(1),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: "Example document",
|
||||||
|
Model: OLD_MODEL,
|
||||||
|
}),
|
||||||
|
Payload: qdrant.NewValueMap(map[string]any{"text": "Example document"}),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{
|
||||||
|
{
|
||||||
|
Id: qdrant.NewIDNum(1),
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: "Example document",
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
Payload: qdrant.NewValueMap(map[string]any{"text": "Example document"}),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
var lastOffset *qdrant.PointId
|
||||||
|
batchSize := uint32(100) // Number of points to read in each batch
|
||||||
|
reachedEnd := false
|
||||||
|
|
||||||
|
for !reachedEnd {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
scrollResult, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Limit: qdrant.PtrOf(batchSize),
|
||||||
|
Offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
WithPayload: qdrant.NewWithPayload(true),
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
WithVectors: qdrant.NewWithVectors(false),
|
||||||
|
})
|
||||||
|
|
||||||
|
records := scrollResult
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
points := make([]*qdrant.PointStruct, len(records))
|
||||||
|
for idx, record := range records {
|
||||||
|
text := ""
|
||||||
|
if val, ok := record.Payload["text"]; ok {
|
||||||
|
text = val.GetStringValue()
|
||||||
|
}
|
||||||
|
|
||||||
|
points[idx] = &qdrant.PointStruct{
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
Id: record.Id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: text,
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
// Keep the original payload
|
||||||
|
Payload: record.Payload,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpdateMode: qdrant.UpdateMode_InsertOnly.Enum(),
|
||||||
|
})
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = (lastOffset == nil)
|
||||||
|
}
|
||||||
|
|
||||||
|
results, err := client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{
|
||||||
|
Text: "my query",
|
||||||
|
Model: OLD_MODEL,
|
||||||
|
}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(10)),
|
||||||
|
})
|
||||||
|
|
||||||
|
results, err = client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{
|
||||||
|
Text: "my query",
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(10)),
|
||||||
|
})
|
||||||
|
```
|
||||||
+144
@@ -0,0 +1,144 @@
|
|||||||
|
```java
|
||||||
|
import static io.qdrant.client.PointIdFactory.id;
|
||||||
|
import static io.qdrant.client.QueryFactory.nearest;
|
||||||
|
import static io.qdrant.client.ValueFactory.value;
|
||||||
|
import static io.qdrant.client.VectorFactory.vector;
|
||||||
|
import static io.qdrant.client.VectorsFactory.vectors;
|
||||||
|
import io.qdrant.client.WithPayloadSelectorFactory;
|
||||||
|
import io.qdrant.client.WithVectorsSelectorFactory;
|
||||||
|
|
||||||
|
import io.qdrant.client.QdrantClient;
|
||||||
|
import io.qdrant.client.QdrantGrpcClient;
|
||||||
|
import io.qdrant.client.grpc.Collections.Distance;
|
||||||
|
import io.qdrant.client.grpc.Collections.VectorParams;
|
||||||
|
import io.qdrant.client.grpc.JsonWithInt.Value;
|
||||||
|
import io.qdrant.client.grpc.Points.Document;
|
||||||
|
import io.qdrant.client.grpc.Points.PointStruct;
|
||||||
|
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.UpsertPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.ScrollPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.UpdateMode;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
client.createCollectionAsync(NEW_COLLECTION,
|
||||||
|
VectorParams.newBuilder()
|
||||||
|
.setSize(512) // Size of the new embedding vectors
|
||||||
|
.setDistance(Distance.Cosine) // Similarity function for the new model
|
||||||
|
.build()).get();
|
||||||
|
|
||||||
|
client.upsertAsync(OLD_COLLECTION, List.of(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(1))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("Example document")
|
||||||
|
.setModel(OLD_MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(Map.of("text", value("Example document")))
|
||||||
|
.build())).get();
|
||||||
|
|
||||||
|
client.upsertAsync(NEW_COLLECTION, List.of(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(1))
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("Example document")
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(Map.of("text", value("Example document")))
|
||||||
|
.build())).get();
|
||||||
|
|
||||||
|
int batchSize = 100; // Number of points to read in each batch
|
||||||
|
boolean reachedEnd = false;
|
||||||
|
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
var scrollBuilder = ScrollPoints.newBuilder()
|
||||||
|
.setCollectionName(OLD_COLLECTION)
|
||||||
|
.setLimit(batchSize)
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.enable(true))
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
.setWithVectors(WithVectorsSelectorFactory.enable(false));
|
||||||
|
|
||||||
|
while (!reachedEnd) {
|
||||||
|
var scrollResult = client.scrollAsync(scrollBuilder.build()).get();
|
||||||
|
|
||||||
|
var records = scrollResult.getResultList();
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (var record : records) {
|
||||||
|
String text = record.getPayloadMap().containsKey("text")
|
||||||
|
? record.getPayloadMap().get("text").getStringValue()
|
||||||
|
: "";
|
||||||
|
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
.setId(record.getId())
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(text)
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build())))
|
||||||
|
// Keep the original payload
|
||||||
|
.putAllPayload(record.getPayloadMap())
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client.upsertAsync(
|
||||||
|
UpsertPoints.newBuilder()
|
||||||
|
.setCollectionName(NEW_COLLECTION)
|
||||||
|
.addAllPoints(points)
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
.setUpdateMode(UpdateMode.InsertOnly)
|
||||||
|
.build()).get();
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
if (scrollResult.hasNextPageOffset()) {
|
||||||
|
scrollBuilder.setOffset(scrollResult.getNextPageOffset());
|
||||||
|
} else {
|
||||||
|
reachedEnd = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
QueryPoints oldRequest =
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(OLD_COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("my query")
|
||||||
|
.setModel(OLD_MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(10)
|
||||||
|
.build();
|
||||||
|
|
||||||
|
var results = client.queryAsync(oldRequest).get();
|
||||||
|
|
||||||
|
QueryPoints newRequest =
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(NEW_COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("my query")
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(10)
|
||||||
|
.build();
|
||||||
|
|
||||||
|
results = client.queryAsync(newRequest).get();
|
||||||
|
```
|
||||||
+60
@@ -0,0 +1,60 @@
|
|||||||
|
```csharp
|
||||||
|
PointId? lastOffset = null;
|
||||||
|
uint limit = 100; // Number of points to read in each batch
|
||||||
|
bool reachedEnd = false;
|
||||||
|
|
||||||
|
while (!reachedEnd)
|
||||||
|
{
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
var scrollResult = await client.ScrollAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
limit: limit,
|
||||||
|
offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
payloadSelector: true,
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
vectorsSelector: false
|
||||||
|
);
|
||||||
|
|
||||||
|
var records = scrollResult.Result;
|
||||||
|
lastOffset = scrollResult.NextPageOffset;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
var points = new List<PointStruct>();
|
||||||
|
foreach (var record in records)
|
||||||
|
{
|
||||||
|
var text = record.Payload.ContainsKey("text")
|
||||||
|
? record.Payload["text"].StringValue
|
||||||
|
: "";
|
||||||
|
|
||||||
|
points.Add(new PointStruct
|
||||||
|
{
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
Id = record.Id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = text,
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
// Keep the original payload
|
||||||
|
Payload = { record.Payload }
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
await client.UpsertAsync(
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
CollectionName = NEW_COLLECTION,
|
||||||
|
Points = { points },
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpdateMode = UpdateMode.InsertOnly
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = (lastOffset == null);
|
||||||
|
}
|
||||||
|
```
|
||||||
+53
@@ -0,0 +1,53 @@
|
|||||||
|
```go
|
||||||
|
var lastOffset *qdrant.PointId
|
||||||
|
batchSize := uint32(100) // Number of points to read in each batch
|
||||||
|
reachedEnd := false
|
||||||
|
|
||||||
|
for !reachedEnd {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
scrollResult, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Limit: qdrant.PtrOf(batchSize),
|
||||||
|
Offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
WithPayload: qdrant.NewWithPayload(true),
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
WithVectors: qdrant.NewWithVectors(false),
|
||||||
|
})
|
||||||
|
|
||||||
|
records := scrollResult
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
points := make([]*qdrant.PointStruct, len(records))
|
||||||
|
for idx, record := range records {
|
||||||
|
text := ""
|
||||||
|
if val, ok := record.Payload["text"]; ok {
|
||||||
|
text = val.GetStringValue()
|
||||||
|
}
|
||||||
|
|
||||||
|
points[idx] = &qdrant.PointStruct{
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
Id: record.Id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: text,
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
// Keep the original payload
|
||||||
|
Payload: record.Payload,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpdateMode: qdrant.UpdateMode_InsertOnly.Enum(),
|
||||||
|
})
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = (lastOffset == nil)
|
||||||
|
}
|
||||||
|
```
|
||||||
+60
@@ -0,0 +1,60 @@
|
|||||||
|
```java
|
||||||
|
int batchSize = 100; // Number of points to read in each batch
|
||||||
|
boolean reachedEnd = false;
|
||||||
|
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
var scrollBuilder = ScrollPoints.newBuilder()
|
||||||
|
.setCollectionName(OLD_COLLECTION)
|
||||||
|
.setLimit(batchSize)
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.enable(true))
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
.setWithVectors(WithVectorsSelectorFactory.enable(false));
|
||||||
|
|
||||||
|
while (!reachedEnd) {
|
||||||
|
var scrollResult = client.scrollAsync(scrollBuilder.build()).get();
|
||||||
|
|
||||||
|
var records = scrollResult.getResultList();
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (var record : records) {
|
||||||
|
String text = record.getPayloadMap().containsKey("text")
|
||||||
|
? record.getPayloadMap().get("text").getStringValue()
|
||||||
|
: "";
|
||||||
|
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
.setId(record.getId())
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(text)
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build())))
|
||||||
|
// Keep the original payload
|
||||||
|
.putAllPayload(record.getPayloadMap())
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client.upsertAsync(
|
||||||
|
UpsertPoints.newBuilder()
|
||||||
|
.setCollectionName(NEW_COLLECTION)
|
||||||
|
.addAllPoints(points)
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
.setUpdateMode(UpdateMode.InsertOnly)
|
||||||
|
.build()).get();
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
if (scrollResult.hasNextPageOffset()) {
|
||||||
|
scrollBuilder.setOffset(scrollResult.getNextPageOffset());
|
||||||
|
} else {
|
||||||
|
reachedEnd = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+45
@@ -0,0 +1,45 @@
|
|||||||
|
```python
|
||||||
|
last_offset = None
|
||||||
|
batch_size = 100 # Number of points to read in each batch
|
||||||
|
reached_end = False
|
||||||
|
|
||||||
|
while not reached_end:
|
||||||
|
# Get the next batch of points from the old collection
|
||||||
|
records, last_offset = client.scroll(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
limit=batch_size,
|
||||||
|
offset=last_offset,
|
||||||
|
# Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
with_payload=True,
|
||||||
|
# We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Re-embed the points using the new model
|
||||||
|
points = [
|
||||||
|
models.PointStruct(
|
||||||
|
# Keep the original ID to ensure consistency
|
||||||
|
id=record.id,
|
||||||
|
# Use the new embedding model to encode the text from the payload,
|
||||||
|
# assuming that was the original source of the embedding
|
||||||
|
vector=models.Document(
|
||||||
|
text=(record.payload or {}).get("text", ""),
|
||||||
|
model=NEW_MODEL,
|
||||||
|
),
|
||||||
|
# Keep the original payload
|
||||||
|
payload=record.payload
|
||||||
|
)
|
||||||
|
for record in records
|
||||||
|
]
|
||||||
|
|
||||||
|
# Upsert the re-embedded points into the new collection
|
||||||
|
client.upsert(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
points=points,
|
||||||
|
# Only insert the point if a point with this ID does not already exist.
|
||||||
|
update_mode=models.UpdateMode.INSERT_ONLY
|
||||||
|
)
|
||||||
|
|
||||||
|
# Check if we reached the end of the collection
|
||||||
|
reached_end = (last_offset == None)
|
||||||
|
```
|
||||||
+58
@@ -0,0 +1,58 @@
|
|||||||
|
```rust
|
||||||
|
let mut last_offset = None;
|
||||||
|
let batch_size = 100; // Number of points to read in each batch
|
||||||
|
|
||||||
|
loop {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
let mut scroll_builder = ScrollPointsBuilder::new(old_collection)
|
||||||
|
.limit(batch_size)
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
.with_payload(true)
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
.with_vectors(false);
|
||||||
|
|
||||||
|
if let Some(offset) = last_offset {
|
||||||
|
scroll_builder = scroll_builder.offset(offset);
|
||||||
|
}
|
||||||
|
|
||||||
|
let scroll_result = client.scroll(scroll_builder).await?;
|
||||||
|
|
||||||
|
let records = scroll_result.result;
|
||||||
|
last_offset = scroll_result.next_page_offset;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
let points: Vec<PointStruct> = records
|
||||||
|
.iter()
|
||||||
|
.map(|record| {
|
||||||
|
PointStruct::new(
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
record.id.clone().unwrap(),
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Document::new(
|
||||||
|
record.payload.get("text")
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
.map_or("", |v| v),
|
||||||
|
new_model,
|
||||||
|
),
|
||||||
|
// Keep the original payload
|
||||||
|
record.payload.clone(),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client
|
||||||
|
.upsert_points(
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpsertPointsBuilder::new(new_collection, points)
|
||||||
|
.update_mode(UpdateMode::InsertOnly),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
if last_offset.is_none() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
+44
@@ -0,0 +1,44 @@
|
|||||||
|
```typescript
|
||||||
|
let lastOffset: number | string | undefined = undefined;
|
||||||
|
const batchSize = 100; // Number of points to read in each batch
|
||||||
|
let reachedEnd = false;
|
||||||
|
|
||||||
|
while (!reachedEnd) {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
const scrollResult = await client.scroll(OLD_COLLECTION, {
|
||||||
|
limit: batchSize,
|
||||||
|
offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
with_payload: true,
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
with_vector: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
const records = scrollResult.points;
|
||||||
|
lastOffset = scrollResult.next_page_offset as number | string | undefined;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
const points = records.map((record) => ({
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
id: record.id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
vector: {
|
||||||
|
text: ((record.payload?.text as string) ?? ""),
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
// Keep the original payload
|
||||||
|
payload: record.payload,
|
||||||
|
}));
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
await client.upsert(NEW_COLLECTION, {
|
||||||
|
points,
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
update_mode: "insert_only" as const,
|
||||||
|
});
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = lastOffset == null;
|
||||||
|
}
|
||||||
|
```
|
||||||
+98
@@ -0,0 +1,98 @@
|
|||||||
|
```python
|
||||||
|
from qdrant_client import QdrantClient, models
|
||||||
|
|
||||||
|
client.create_collection(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
vectors_config=(
|
||||||
|
models.VectorParams(
|
||||||
|
size=512, # Size of the new embedding vectors
|
||||||
|
distance=models.Distance.COSINE # Similarity function for the new model
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
client.upsert(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=1,
|
||||||
|
vector=models.Document(
|
||||||
|
text="Example document",
|
||||||
|
model=OLD_MODEL,
|
||||||
|
),
|
||||||
|
payload={"text": "Example document"}
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
client.upsert(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=1,
|
||||||
|
# Use the new embedding model to encode the document
|
||||||
|
vector=models.Document(
|
||||||
|
text="Example document",
|
||||||
|
model=NEW_MODEL,
|
||||||
|
),
|
||||||
|
payload={"text": "Example document"}
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
last_offset = None
|
||||||
|
batch_size = 100 # Number of points to read in each batch
|
||||||
|
reached_end = False
|
||||||
|
|
||||||
|
while not reached_end:
|
||||||
|
# Get the next batch of points from the old collection
|
||||||
|
records, last_offset = client.scroll(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
limit=batch_size,
|
||||||
|
offset=last_offset,
|
||||||
|
# Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
with_payload=True,
|
||||||
|
# We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Re-embed the points using the new model
|
||||||
|
points = [
|
||||||
|
models.PointStruct(
|
||||||
|
# Keep the original ID to ensure consistency
|
||||||
|
id=record.id,
|
||||||
|
# Use the new embedding model to encode the text from the payload,
|
||||||
|
# assuming that was the original source of the embedding
|
||||||
|
vector=models.Document(
|
||||||
|
text=(record.payload or {}).get("text", ""),
|
||||||
|
model=NEW_MODEL,
|
||||||
|
),
|
||||||
|
# Keep the original payload
|
||||||
|
payload=record.payload
|
||||||
|
)
|
||||||
|
for record in records
|
||||||
|
]
|
||||||
|
|
||||||
|
# Upsert the re-embedded points into the new collection
|
||||||
|
client.upsert(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
points=points,
|
||||||
|
# Only insert the point if a point with this ID does not already exist.
|
||||||
|
update_mode=models.UpdateMode.INSERT_ONLY
|
||||||
|
)
|
||||||
|
|
||||||
|
# Check if we reached the end of the collection
|
||||||
|
reached_end = (last_offset == None)
|
||||||
|
|
||||||
|
results = client.query_points(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
query=models.Document(text="my query", model=OLD_MODEL),
|
||||||
|
limit=10,
|
||||||
|
)
|
||||||
|
|
||||||
|
results = client.query_points(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
query=models.Document(text="my query", model=NEW_MODEL),
|
||||||
|
limit=10,
|
||||||
|
)
|
||||||
|
```
|
||||||
+110
@@ -0,0 +1,110 @@
|
|||||||
|
```rust
|
||||||
|
use qdrant_client::qdrant::{
|
||||||
|
CreateCollectionBuilder, Distance, Document, PointStruct, Query, QueryPointsBuilder,
|
||||||
|
ScrollPointsBuilder, UpdateMode, UpsertPointsBuilder, VectorParamsBuilder,
|
||||||
|
};
|
||||||
|
use qdrant_client::Qdrant;
|
||||||
|
|
||||||
|
client
|
||||||
|
.create_collection(
|
||||||
|
CreateCollectionBuilder::new(new_collection)
|
||||||
|
.vectors_config(VectorParamsBuilder::new(512, Distance::Cosine)), // Size of the new embedding vectors
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(
|
||||||
|
old_collection,
|
||||||
|
vec![PointStruct::new(
|
||||||
|
1,
|
||||||
|
Document::new("Example document", old_model),
|
||||||
|
[("text", "Example document".into())],
|
||||||
|
)],
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(
|
||||||
|
new_collection,
|
||||||
|
vec![PointStruct::new(
|
||||||
|
1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Document::new("Example document", new_model),
|
||||||
|
[("text", "Example document".into())],
|
||||||
|
)],
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let mut last_offset = None;
|
||||||
|
let batch_size = 100; // Number of points to read in each batch
|
||||||
|
|
||||||
|
loop {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
let mut scroll_builder = ScrollPointsBuilder::new(old_collection)
|
||||||
|
.limit(batch_size)
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
.with_payload(true)
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
.with_vectors(false);
|
||||||
|
|
||||||
|
if let Some(offset) = last_offset {
|
||||||
|
scroll_builder = scroll_builder.offset(offset);
|
||||||
|
}
|
||||||
|
|
||||||
|
let scroll_result = client.scroll(scroll_builder).await?;
|
||||||
|
|
||||||
|
let records = scroll_result.result;
|
||||||
|
last_offset = scroll_result.next_page_offset;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
let points: Vec<PointStruct> = records
|
||||||
|
.iter()
|
||||||
|
.map(|record| {
|
||||||
|
PointStruct::new(
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
record.id.clone().unwrap(),
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Document::new(
|
||||||
|
record.payload.get("text")
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
.map_or("", |v| v),
|
||||||
|
new_model,
|
||||||
|
),
|
||||||
|
// Keep the original payload
|
||||||
|
record.payload.clone(),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client
|
||||||
|
.upsert_points(
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpsertPointsBuilder::new(new_collection, points)
|
||||||
|
.update_mode(UpdateMode::InsertOnly),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
if last_offset.is_none() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let results = client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(old_collection)
|
||||||
|
.query(Query::new_nearest(Document::new("my query", old_model)))
|
||||||
|
.limit(10),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let results = client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(new_collection)
|
||||||
|
.query(Query::new_nearest(Document::new("my query", new_model)))
|
||||||
|
.limit(10),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
```csharp
|
||||||
|
results = await client.QueryAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
query: new Document
|
||||||
|
{
|
||||||
|
Text = "my query",
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
limit: 10
|
||||||
|
);
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```go
|
||||||
|
results, err = client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{
|
||||||
|
Text: "my query",
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(10)),
|
||||||
|
})
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```java
|
||||||
|
QueryPoints newRequest =
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(NEW_COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("my query")
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(10)
|
||||||
|
.build();
|
||||||
|
|
||||||
|
results = client.queryAsync(newRequest).get();
|
||||||
|
```
|
||||||
+7
@@ -0,0 +1,7 @@
|
|||||||
|
```python
|
||||||
|
results = client.query_points(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
query=models.Document(text="my query", model=NEW_MODEL),
|
||||||
|
limit=10,
|
||||||
|
)
|
||||||
|
```
|
||||||
+9
@@ -0,0 +1,9 @@
|
|||||||
|
```rust
|
||||||
|
let results = client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(new_collection)
|
||||||
|
.query(Query::new_nearest(Document::new("my query", new_model)))
|
||||||
|
.limit(10),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+9
@@ -0,0 +1,9 @@
|
|||||||
|
```typescript
|
||||||
|
const resultsNew = await client.query(NEW_COLLECTION, {
|
||||||
|
query: {
|
||||||
|
text: "my query",
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
limit: 10,
|
||||||
|
});
|
||||||
|
```
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
```csharp
|
||||||
|
var results = await client.QueryAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
query: new Document
|
||||||
|
{
|
||||||
|
Text = "my query",
|
||||||
|
Model = OLD_MODEL
|
||||||
|
},
|
||||||
|
limit: 10
|
||||||
|
);
|
||||||
|
```
|
||||||
+10
@@ -0,0 +1,10 @@
|
|||||||
|
```go
|
||||||
|
results, err := client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{
|
||||||
|
Text: "my query",
|
||||||
|
Model: OLD_MODEL,
|
||||||
|
}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(10)),
|
||||||
|
})
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```java
|
||||||
|
QueryPoints oldRequest =
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(OLD_COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("my query")
|
||||||
|
.setModel(OLD_MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(10)
|
||||||
|
.build();
|
||||||
|
|
||||||
|
var results = client.queryAsync(oldRequest).get();
|
||||||
|
```
|
||||||
+7
@@ -0,0 +1,7 @@
|
|||||||
|
```python
|
||||||
|
results = client.query_points(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
query=models.Document(text="my query", model=OLD_MODEL),
|
||||||
|
limit=10,
|
||||||
|
)
|
||||||
|
```
|
||||||
+9
@@ -0,0 +1,9 @@
|
|||||||
|
```rust
|
||||||
|
let results = client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(old_collection)
|
||||||
|
.query(Query::new_nearest(Document::new("my query", old_model)))
|
||||||
|
.limit(10),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+9
@@ -0,0 +1,9 @@
|
|||||||
|
```typescript
|
||||||
|
const results = await client.query(OLD_COLLECTION, {
|
||||||
|
query: {
|
||||||
|
text: "my query",
|
||||||
|
model: OLD_MODEL,
|
||||||
|
},
|
||||||
|
limit: 10,
|
||||||
|
});
|
||||||
|
```
|
||||||
+96
@@ -0,0 +1,96 @@
|
|||||||
|
```typescript
|
||||||
|
import { QdrantClient } from "@qdrant/js-client-rest";
|
||||||
|
|
||||||
|
await client.createCollection(NEW_COLLECTION, {
|
||||||
|
vectors: {
|
||||||
|
size: 512, // Size of the new embedding vectors
|
||||||
|
distance: "Cosine", // Similarity function for the new model
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
await client.upsert(OLD_COLLECTION, {
|
||||||
|
points: [
|
||||||
|
{
|
||||||
|
id: 1,
|
||||||
|
vector: {
|
||||||
|
text: "Example document",
|
||||||
|
model: OLD_MODEL,
|
||||||
|
},
|
||||||
|
payload: { text: "Example document" },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
});
|
||||||
|
|
||||||
|
await client.upsert(NEW_COLLECTION, {
|
||||||
|
points: [
|
||||||
|
{
|
||||||
|
id: 1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
vector: {
|
||||||
|
text: "Example document",
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
payload: { text: "Example document" },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
});
|
||||||
|
|
||||||
|
let lastOffset: number | string | undefined = undefined;
|
||||||
|
const batchSize = 100; // Number of points to read in each batch
|
||||||
|
let reachedEnd = false;
|
||||||
|
|
||||||
|
while (!reachedEnd) {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
const scrollResult = await client.scroll(OLD_COLLECTION, {
|
||||||
|
limit: batchSize,
|
||||||
|
offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
with_payload: true,
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
with_vector: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
const records = scrollResult.points;
|
||||||
|
lastOffset = scrollResult.next_page_offset as number | string | undefined;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
const points = records.map((record) => ({
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
id: record.id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
vector: {
|
||||||
|
text: ((record.payload?.text as string) ?? ""),
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
// Keep the original payload
|
||||||
|
payload: record.payload,
|
||||||
|
}));
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
await client.upsert(NEW_COLLECTION, {
|
||||||
|
points,
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
update_mode: "insert_only" as const,
|
||||||
|
});
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = lastOffset == null;
|
||||||
|
}
|
||||||
|
|
||||||
|
const results = await client.query(OLD_COLLECTION, {
|
||||||
|
query: {
|
||||||
|
text: "my query",
|
||||||
|
model: OLD_MODEL,
|
||||||
|
},
|
||||||
|
limit: 10,
|
||||||
|
});
|
||||||
|
|
||||||
|
const resultsNew = await client.query(NEW_COLLECTION, {
|
||||||
|
query: {
|
||||||
|
text: "my query",
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
limit: 10,
|
||||||
|
});
|
||||||
|
```
|
||||||
+19
@@ -0,0 +1,19 @@
|
|||||||
|
```csharp
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: NEW_COLLECTION,
|
||||||
|
points: new List<PointStruct>
|
||||||
|
{
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
Id = 1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = "Example document",
|
||||||
|
Model = NEW_MODEL
|
||||||
|
},
|
||||||
|
Payload = { ["text"] = "Example document" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
);
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```go
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{
|
||||||
|
{
|
||||||
|
Id: qdrant.NewIDNum(1),
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: "Example document",
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
Payload: qdrant.NewValueMap(map[string]any{"text": "Example document"}),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```java
|
||||||
|
client.upsertAsync(NEW_COLLECTION, List.of(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(1))
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("Example document")
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(Map.of("text", value("Example document")))
|
||||||
|
.build())).get();
|
||||||
|
```
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
```python
|
||||||
|
client.upsert(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=1,
|
||||||
|
# Use the new embedding model to encode the document
|
||||||
|
vector=models.Document(
|
||||||
|
text="Example document",
|
||||||
|
model=NEW_MODEL,
|
||||||
|
),
|
||||||
|
payload={"text": "Example document"}
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
```
|
||||||
+13
@@ -0,0 +1,13 @@
|
|||||||
|
```rust
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(
|
||||||
|
new_collection,
|
||||||
|
vec![PointStruct::new(
|
||||||
|
1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Document::new("Example document", new_model),
|
||||||
|
[("text", "Example document".into())],
|
||||||
|
)],
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```typescript
|
||||||
|
await client.upsert(NEW_COLLECTION, {
|
||||||
|
points: [
|
||||||
|
{
|
||||||
|
id: 1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
vector: {
|
||||||
|
text: "Example document",
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
payload: { text: "Example document" },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
});
|
||||||
|
```
|
||||||
+18
@@ -0,0 +1,18 @@
|
|||||||
|
```csharp
|
||||||
|
await client.UpsertAsync(
|
||||||
|
collectionName: OLD_COLLECTION,
|
||||||
|
points: new List<PointStruct>
|
||||||
|
{
|
||||||
|
new()
|
||||||
|
{
|
||||||
|
Id = 1,
|
||||||
|
Vectors = new Document
|
||||||
|
{
|
||||||
|
Text = "Example document",
|
||||||
|
Model = OLD_MODEL
|
||||||
|
},
|
||||||
|
Payload = { ["text"] = "Example document" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
);
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```go
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{
|
||||||
|
{
|
||||||
|
Id: qdrant.NewIDNum(1),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: "Example document",
|
||||||
|
Model: OLD_MODEL,
|
||||||
|
}),
|
||||||
|
Payload: qdrant.NewValueMap(map[string]any{"text": "Example document"}),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
```
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
```java
|
||||||
|
client.upsertAsync(OLD_COLLECTION, List.of(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(1))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("Example document")
|
||||||
|
.setModel(OLD_MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(Map.of("text", value("Example document")))
|
||||||
|
.build())).get();
|
||||||
|
```
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
```python
|
||||||
|
client.upsert(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=1,
|
||||||
|
vector=models.Document(
|
||||||
|
text="Example document",
|
||||||
|
model=OLD_MODEL,
|
||||||
|
),
|
||||||
|
payload={"text": "Example document"}
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
```
|
||||||
+12
@@ -0,0 +1,12 @@
|
|||||||
|
```rust
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(
|
||||||
|
old_collection,
|
||||||
|
vec![PointStruct::new(
|
||||||
|
1,
|
||||||
|
Document::new("Example document", old_model),
|
||||||
|
[("text", "Example document".into())],
|
||||||
|
)],
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
```
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
```typescript
|
||||||
|
await client.upsert(OLD_COLLECTION, {
|
||||||
|
points: [
|
||||||
|
{
|
||||||
|
id: 1,
|
||||||
|
vector: {
|
||||||
|
text: "Example document",
|
||||||
|
model: OLD_MODEL,
|
||||||
|
},
|
||||||
|
payload: { text: "Example document" },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
});
|
||||||
|
```
|
||||||
+167
@@ -0,0 +1,167 @@
|
|||||||
|
package snippet
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
|
||||||
|
"github.com/qdrant/go-client/qdrant"
|
||||||
|
)
|
||||||
|
|
||||||
|
func Main() {
|
||||||
|
// @hide-start
|
||||||
|
client, err := qdrant.NewClient(&qdrant.Config{
|
||||||
|
Host: "",
|
||||||
|
APIKey: "",
|
||||||
|
UseTLS: true,
|
||||||
|
})
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
NEW_COLLECTION := "new_collection"
|
||||||
|
OLD_COLLECTION := "old_collection"
|
||||||
|
|
||||||
|
OLD_MODEL := "sentence-transformers/all-minilm-l6-v2"
|
||||||
|
NEW_MODEL := "qdrant/clip-vit-b-32-text"
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-new-collection
|
||||||
|
client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
|
||||||
|
Size: 512, // Size of the new embedding vectors
|
||||||
|
Distance: qdrant.Distance_Cosine,
|
||||||
|
}),
|
||||||
|
})
|
||||||
|
// @block-end create-new-collection
|
||||||
|
|
||||||
|
// @block-start upsert-old-collection
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{
|
||||||
|
{
|
||||||
|
Id: qdrant.NewIDNum(1),
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: "Example document",
|
||||||
|
Model: OLD_MODEL,
|
||||||
|
}),
|
||||||
|
Payload: qdrant.NewValueMap(map[string]any{"text": "Example document"}),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
// @block-end upsert-old-collection
|
||||||
|
|
||||||
|
// @block-start upsert-new-collection
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Points: []*qdrant.PointStruct{
|
||||||
|
{
|
||||||
|
Id: qdrant.NewIDNum(1),
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: "Example document",
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
Payload: qdrant.NewValueMap(map[string]any{"text": "Example document"}),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
// @block-end upsert-new-collection
|
||||||
|
|
||||||
|
// @block-start migrate-points
|
||||||
|
var lastOffset *qdrant.PointId
|
||||||
|
batchSize := uint32(100) // Number of points to read in each batch
|
||||||
|
reachedEnd := false
|
||||||
|
|
||||||
|
for !reachedEnd {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
scrollResult, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Limit: qdrant.PtrOf(batchSize),
|
||||||
|
Offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
WithPayload: qdrant.NewWithPayload(true),
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
WithVectors: qdrant.NewWithVectors(false),
|
||||||
|
})
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
if err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
records := scrollResult
|
||||||
|
lastOffset = scrollResult[len(scrollResult)-1].Id // @hide
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
points := make([]*qdrant.PointStruct, len(records))
|
||||||
|
for idx, record := range records {
|
||||||
|
text := ""
|
||||||
|
if val, ok := record.Payload["text"]; ok {
|
||||||
|
text = val.GetStringValue()
|
||||||
|
}
|
||||||
|
|
||||||
|
points[idx] = &qdrant.PointStruct{
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
Id: record.Id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Vectors: qdrant.NewVectorsDocument(&qdrant.Document{
|
||||||
|
Text: text,
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
// Keep the original payload
|
||||||
|
Payload: record.Payload,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client.Upsert(context.Background(), &qdrant.UpsertPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Points: points,
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpdateMode: qdrant.UpdateMode_InsertOnly.Enum(),
|
||||||
|
})
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = (lastOffset == nil)
|
||||||
|
}
|
||||||
|
// @block-end migrate-points
|
||||||
|
|
||||||
|
// @block-start search-old-collection
|
||||||
|
results, err := client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: OLD_COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{
|
||||||
|
Text: "my query",
|
||||||
|
Model: OLD_MODEL,
|
||||||
|
}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(10)),
|
||||||
|
})
|
||||||
|
// @block-end search-old-collection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
if err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
_ = results
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start search-new-collection
|
||||||
|
results, err = client.Query(context.Background(), &qdrant.QueryPoints{
|
||||||
|
CollectionName: NEW_COLLECTION,
|
||||||
|
Query: qdrant.NewQueryDocument(&qdrant.Document{
|
||||||
|
Text: "my query",
|
||||||
|
Model: NEW_MODEL,
|
||||||
|
}),
|
||||||
|
Limit: qdrant.PtrOf(uint64(10)),
|
||||||
|
})
|
||||||
|
// @block-end search-new-collection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
if err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
_ = results
|
||||||
|
// @hide-end
|
||||||
|
}
|
||||||
+179
@@ -0,0 +1,179 @@
|
|||||||
|
package com.example.snippets_amalgamation;
|
||||||
|
|
||||||
|
import static io.qdrant.client.PointIdFactory.id;
|
||||||
|
import static io.qdrant.client.QueryFactory.nearest;
|
||||||
|
import static io.qdrant.client.ValueFactory.value;
|
||||||
|
import static io.qdrant.client.VectorFactory.vector;
|
||||||
|
import static io.qdrant.client.VectorsFactory.vectors;
|
||||||
|
import io.qdrant.client.WithPayloadSelectorFactory;
|
||||||
|
import io.qdrant.client.WithVectorsSelectorFactory;
|
||||||
|
|
||||||
|
import io.qdrant.client.QdrantClient;
|
||||||
|
import io.qdrant.client.QdrantGrpcClient;
|
||||||
|
import io.qdrant.client.grpc.Collections.Distance;
|
||||||
|
import io.qdrant.client.grpc.Collections.VectorParams;
|
||||||
|
import io.qdrant.client.grpc.JsonWithInt.Value;
|
||||||
|
import io.qdrant.client.grpc.Points.Document;
|
||||||
|
import io.qdrant.client.grpc.Points.PointStruct;
|
||||||
|
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.UpsertPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.ScrollPoints;
|
||||||
|
import io.qdrant.client.grpc.Points.UpdateMode;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
public class Snippet {
|
||||||
|
|
||||||
|
public static void run() throws Exception {
|
||||||
|
// @hide-start
|
||||||
|
String QDRANT_URL = "";
|
||||||
|
String QDRANT_API_KEY = "";
|
||||||
|
|
||||||
|
QdrantClient client =
|
||||||
|
new QdrantClient(
|
||||||
|
QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
|
||||||
|
.withApiKey(QDRANT_API_KEY)
|
||||||
|
.build());
|
||||||
|
|
||||||
|
String NEW_COLLECTION = "new_collection";
|
||||||
|
String OLD_COLLECTION = "old_collection";
|
||||||
|
|
||||||
|
String OLD_MODEL = "sentence-transformers/all-minilm-l6-v2";
|
||||||
|
String NEW_MODEL = "qdrant/clip-vit-b-32-text";
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-new-collection
|
||||||
|
client.createCollectionAsync(NEW_COLLECTION,
|
||||||
|
VectorParams.newBuilder()
|
||||||
|
.setSize(512) // Size of the new embedding vectors
|
||||||
|
.setDistance(Distance.Cosine) // Similarity function for the new model
|
||||||
|
.build()).get();
|
||||||
|
// @block-end create-new-collection
|
||||||
|
|
||||||
|
// @block-start upsert-old-collection
|
||||||
|
client.upsertAsync(OLD_COLLECTION, List.of(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(1))
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("Example document")
|
||||||
|
.setModel(OLD_MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(Map.of("text", value("Example document")))
|
||||||
|
.build())).get();
|
||||||
|
// @block-end upsert-old-collection
|
||||||
|
|
||||||
|
// @block-start upsert-new-collection
|
||||||
|
client.upsertAsync(NEW_COLLECTION, List.of(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
.setId(id(1))
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("Example document")
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build())))
|
||||||
|
.putAllPayload(Map.of("text", value("Example document")))
|
||||||
|
.build())).get();
|
||||||
|
// @block-end upsert-new-collection
|
||||||
|
|
||||||
|
// @block-start migrate-points
|
||||||
|
int batchSize = 100; // Number of points to read in each batch
|
||||||
|
boolean reachedEnd = false;
|
||||||
|
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
var scrollBuilder = ScrollPoints.newBuilder()
|
||||||
|
.setCollectionName(OLD_COLLECTION)
|
||||||
|
.setLimit(batchSize)
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
.setWithPayload(WithPayloadSelectorFactory.enable(true))
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
.setWithVectors(WithVectorsSelectorFactory.enable(false));
|
||||||
|
|
||||||
|
while (!reachedEnd) {
|
||||||
|
var scrollResult = client.scrollAsync(scrollBuilder.build()).get();
|
||||||
|
|
||||||
|
var records = scrollResult.getResultList();
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
List<PointStruct> points = new ArrayList<>();
|
||||||
|
for (var record : records) {
|
||||||
|
String text = record.getPayloadMap().containsKey("text")
|
||||||
|
? record.getPayloadMap().get("text").getStringValue()
|
||||||
|
: "";
|
||||||
|
|
||||||
|
points.add(
|
||||||
|
PointStruct.newBuilder()
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
.setId(record.getId())
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
.setVectors(
|
||||||
|
vectors(
|
||||||
|
vector(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText(text)
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build())))
|
||||||
|
// Keep the original payload
|
||||||
|
.putAllPayload(record.getPayloadMap())
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client.upsertAsync(
|
||||||
|
UpsertPoints.newBuilder()
|
||||||
|
.setCollectionName(NEW_COLLECTION)
|
||||||
|
.addAllPoints(points)
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
.setUpdateMode(UpdateMode.InsertOnly)
|
||||||
|
.build()).get();
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
if (scrollResult.hasNextPageOffset()) {
|
||||||
|
scrollBuilder.setOffset(scrollResult.getNextPageOffset());
|
||||||
|
} else {
|
||||||
|
reachedEnd = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end migrate-points
|
||||||
|
|
||||||
|
// @block-start search-old-collection
|
||||||
|
QueryPoints oldRequest =
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(OLD_COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("my query")
|
||||||
|
.setModel(OLD_MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(10)
|
||||||
|
.build();
|
||||||
|
|
||||||
|
var results = client.queryAsync(oldRequest).get();
|
||||||
|
// @block-end search-old-collection
|
||||||
|
|
||||||
|
// @block-start search-new-collection
|
||||||
|
QueryPoints newRequest =
|
||||||
|
QueryPoints.newBuilder()
|
||||||
|
.setCollectionName(NEW_COLLECTION)
|
||||||
|
.setQuery(
|
||||||
|
nearest(
|
||||||
|
Document.newBuilder()
|
||||||
|
.setText("my query")
|
||||||
|
.setModel(NEW_MODEL)
|
||||||
|
.build()))
|
||||||
|
.setLimit(10)
|
||||||
|
.build();
|
||||||
|
|
||||||
|
results = client.queryAsync(newRequest).get();
|
||||||
|
// @block-end search-new-collection
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
+121
@@ -0,0 +1,121 @@
|
|||||||
|
from qdrant_client import QdrantClient, models
|
||||||
|
|
||||||
|
# @hide-start
|
||||||
|
client = QdrantClient(
|
||||||
|
url="",
|
||||||
|
api_key=""
|
||||||
|
)
|
||||||
|
|
||||||
|
NEW_COLLECTION="new_collection"
|
||||||
|
OLD_COLLECTION="old_collection"
|
||||||
|
|
||||||
|
OLD_MODEL="sentence-transformers/all-minilm-l6-v2"
|
||||||
|
NEW_MODEL="qdrant/clip-vit-b-32-text"
|
||||||
|
# @hide-end
|
||||||
|
|
||||||
|
# @block-start create-new-collection
|
||||||
|
client.create_collection(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
vectors_config=(
|
||||||
|
models.VectorParams(
|
||||||
|
size=512, # Size of the new embedding vectors
|
||||||
|
distance=models.Distance.COSINE # Similarity function for the new model
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
# @block-end create-new-collection
|
||||||
|
|
||||||
|
# @block-start upsert-old-collection
|
||||||
|
client.upsert(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=1,
|
||||||
|
vector=models.Document(
|
||||||
|
text="Example document",
|
||||||
|
model=OLD_MODEL,
|
||||||
|
),
|
||||||
|
payload={"text": "Example document"}
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
# @block-end upsert-old-collection
|
||||||
|
|
||||||
|
# @block-start upsert-new-collection
|
||||||
|
client.upsert(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
points=[
|
||||||
|
models.PointStruct(
|
||||||
|
id=1,
|
||||||
|
# Use the new embedding model to encode the document
|
||||||
|
vector=models.Document(
|
||||||
|
text="Example document",
|
||||||
|
model=NEW_MODEL,
|
||||||
|
),
|
||||||
|
payload={"text": "Example document"}
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
# @block-end upsert-new-collection
|
||||||
|
|
||||||
|
# @block-start migrate-points
|
||||||
|
last_offset = None
|
||||||
|
batch_size = 100 # Number of points to read in each batch
|
||||||
|
reached_end = False
|
||||||
|
|
||||||
|
while not reached_end:
|
||||||
|
# Get the next batch of points from the old collection
|
||||||
|
records, last_offset = client.scroll(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
limit=batch_size,
|
||||||
|
offset=last_offset,
|
||||||
|
# Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
with_payload=True,
|
||||||
|
# We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Re-embed the points using the new model
|
||||||
|
points = [
|
||||||
|
models.PointStruct(
|
||||||
|
# Keep the original ID to ensure consistency
|
||||||
|
id=record.id,
|
||||||
|
# Use the new embedding model to encode the text from the payload,
|
||||||
|
# assuming that was the original source of the embedding
|
||||||
|
vector=models.Document(
|
||||||
|
text=(record.payload or {}).get("text", ""),
|
||||||
|
model=NEW_MODEL,
|
||||||
|
),
|
||||||
|
# Keep the original payload
|
||||||
|
payload=record.payload
|
||||||
|
)
|
||||||
|
for record in records
|
||||||
|
]
|
||||||
|
|
||||||
|
# Upsert the re-embedded points into the new collection
|
||||||
|
client.upsert(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
points=points,
|
||||||
|
# Only insert the point if a point with this ID does not already exist.
|
||||||
|
update_mode=models.UpdateMode.INSERT_ONLY
|
||||||
|
)
|
||||||
|
|
||||||
|
# Check if we reached the end of the collection
|
||||||
|
reached_end = (last_offset == None)
|
||||||
|
# @block-end migrate-points
|
||||||
|
|
||||||
|
# @block-start search-old-collection
|
||||||
|
results = client.query_points(
|
||||||
|
collection_name=OLD_COLLECTION,
|
||||||
|
query=models.Document(text="my query", model=OLD_MODEL),
|
||||||
|
limit=10,
|
||||||
|
)
|
||||||
|
# @block-end search-old-collection
|
||||||
|
|
||||||
|
# @block-start search-new-collection
|
||||||
|
results = client.query_points(
|
||||||
|
collection_name=NEW_COLLECTION,
|
||||||
|
query=models.Document(text="my query", model=NEW_MODEL),
|
||||||
|
limit=10,
|
||||||
|
)
|
||||||
|
# @block-end search-new-collection
|
||||||
+147
@@ -0,0 +1,147 @@
|
|||||||
|
use qdrant_client::qdrant::{
|
||||||
|
CreateCollectionBuilder, Distance, Document, PointStruct, Query, QueryPointsBuilder,
|
||||||
|
ScrollPointsBuilder, UpdateMode, UpsertPointsBuilder, VectorParamsBuilder,
|
||||||
|
};
|
||||||
|
use qdrant_client::Qdrant;
|
||||||
|
|
||||||
|
pub async fn main() -> anyhow::Result<()> {
|
||||||
|
// @hide-start
|
||||||
|
let QDRANT_URL = "";
|
||||||
|
let QDRANT_API_KEY = "";
|
||||||
|
|
||||||
|
let client = Qdrant::from_url(QDRANT_URL)
|
||||||
|
.api_key(QDRANT_API_KEY)
|
||||||
|
.build()?;
|
||||||
|
|
||||||
|
let new_collection = "new_collection";
|
||||||
|
let old_collection = "old_collection";
|
||||||
|
|
||||||
|
let old_model = "sentence-transformers/all-minilm-l6-v2";
|
||||||
|
let new_model = "qdrant/clip-vit-b-32-text";
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-new-collection
|
||||||
|
client
|
||||||
|
.create_collection(
|
||||||
|
CreateCollectionBuilder::new(new_collection)
|
||||||
|
.vectors_config(VectorParamsBuilder::new(512, Distance::Cosine)), // Size of the new embedding vectors
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
// @block-end create-new-collection
|
||||||
|
|
||||||
|
// @block-start upsert-old-collection
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(
|
||||||
|
old_collection,
|
||||||
|
vec![PointStruct::new(
|
||||||
|
1,
|
||||||
|
Document::new("Example document", old_model),
|
||||||
|
[("text", "Example document".into())],
|
||||||
|
)],
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
// @block-end upsert-old-collection
|
||||||
|
|
||||||
|
// @block-start upsert-new-collection
|
||||||
|
client
|
||||||
|
.upsert_points(UpsertPointsBuilder::new(
|
||||||
|
new_collection,
|
||||||
|
vec![PointStruct::new(
|
||||||
|
1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
Document::new("Example document", new_model),
|
||||||
|
[("text", "Example document".into())],
|
||||||
|
)],
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
// @block-end upsert-new-collection
|
||||||
|
|
||||||
|
// @block-start migrate-points
|
||||||
|
let mut last_offset = None;
|
||||||
|
let batch_size = 100; // Number of points to read in each batch
|
||||||
|
|
||||||
|
loop {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
let mut scroll_builder = ScrollPointsBuilder::new(old_collection)
|
||||||
|
.limit(batch_size)
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
.with_payload(true)
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
.with_vectors(false);
|
||||||
|
|
||||||
|
if let Some(offset) = last_offset {
|
||||||
|
scroll_builder = scroll_builder.offset(offset);
|
||||||
|
}
|
||||||
|
|
||||||
|
let scroll_result = client.scroll(scroll_builder).await?;
|
||||||
|
|
||||||
|
let records = scroll_result.result;
|
||||||
|
last_offset = scroll_result.next_page_offset;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
let points: Vec<PointStruct> = records
|
||||||
|
.iter()
|
||||||
|
.map(|record| {
|
||||||
|
PointStruct::new(
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
record.id.clone().unwrap(),
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
Document::new(
|
||||||
|
record.payload.get("text")
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
.map_or("", |v| v),
|
||||||
|
new_model,
|
||||||
|
),
|
||||||
|
// Keep the original payload
|
||||||
|
record.payload.clone(),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
client
|
||||||
|
.upsert_points(
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
UpsertPointsBuilder::new(new_collection, points)
|
||||||
|
.update_mode(UpdateMode::InsertOnly),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
if last_offset.is_none() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// @block-end migrate-points
|
||||||
|
|
||||||
|
// @block-start search-old-collection
|
||||||
|
let results = client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(old_collection)
|
||||||
|
.query(Query::new_nearest(Document::new("my query", old_model)))
|
||||||
|
.limit(10),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
// @block-end search-old-collection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
_ = results;
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start search-new-collection
|
||||||
|
let results = client
|
||||||
|
.query(
|
||||||
|
QueryPointsBuilder::new(new_collection)
|
||||||
|
.query(Query::new_nearest(Document::new("my query", new_model)))
|
||||||
|
.limit(10),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
// @block-end search-new-collection
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
_ = results;
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
+122
@@ -0,0 +1,122 @@
|
|||||||
|
import { QdrantClient } from "@qdrant/js-client-rest";
|
||||||
|
|
||||||
|
// @hide-start
|
||||||
|
const QDRANT_URL = "";
|
||||||
|
const QDRANT_API_KEY = "";
|
||||||
|
|
||||||
|
const client = new QdrantClient({
|
||||||
|
url: QDRANT_URL,
|
||||||
|
apiKey: QDRANT_API_KEY,
|
||||||
|
});
|
||||||
|
|
||||||
|
const NEW_COLLECTION = "new_collection";
|
||||||
|
const OLD_COLLECTION = "old_collection";
|
||||||
|
|
||||||
|
const OLD_MODEL = "sentence-transformers/all-minilm-l6-v2";
|
||||||
|
const NEW_MODEL = "qdrant/clip-vit-b-32-text";
|
||||||
|
// @hide-end
|
||||||
|
|
||||||
|
// @block-start create-new-collection
|
||||||
|
await client.createCollection(NEW_COLLECTION, {
|
||||||
|
vectors: {
|
||||||
|
size: 512, // Size of the new embedding vectors
|
||||||
|
distance: "Cosine", // Similarity function for the new model
|
||||||
|
},
|
||||||
|
});
|
||||||
|
// @block-end create-new-collection
|
||||||
|
|
||||||
|
// @block-start upsert-old-collection
|
||||||
|
await client.upsert(OLD_COLLECTION, {
|
||||||
|
points: [
|
||||||
|
{
|
||||||
|
id: 1,
|
||||||
|
vector: {
|
||||||
|
text: "Example document",
|
||||||
|
model: OLD_MODEL,
|
||||||
|
},
|
||||||
|
payload: { text: "Example document" },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
});
|
||||||
|
// @block-end upsert-old-collection
|
||||||
|
|
||||||
|
// @block-start upsert-new-collection
|
||||||
|
await client.upsert(NEW_COLLECTION, {
|
||||||
|
points: [
|
||||||
|
{
|
||||||
|
id: 1,
|
||||||
|
// Use the new embedding model to encode the document
|
||||||
|
vector: {
|
||||||
|
text: "Example document",
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
payload: { text: "Example document" },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
});
|
||||||
|
// @block-end upsert-new-collection
|
||||||
|
|
||||||
|
// @block-start migrate-points
|
||||||
|
let lastOffset: number | string | undefined = undefined;
|
||||||
|
const batchSize = 100; // Number of points to read in each batch
|
||||||
|
let reachedEnd = false;
|
||||||
|
|
||||||
|
while (!reachedEnd) {
|
||||||
|
// Get the next batch of points from the old collection
|
||||||
|
const scrollResult = await client.scroll(OLD_COLLECTION, {
|
||||||
|
limit: batchSize,
|
||||||
|
offset: lastOffset,
|
||||||
|
// Include payloads in the response, as we need them to re-embed the vectors
|
||||||
|
with_payload: true,
|
||||||
|
// We don't need the old vectors, so let's save on the bandwidth
|
||||||
|
with_vector: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
const records = scrollResult.points;
|
||||||
|
lastOffset = scrollResult.next_page_offset as number | string | undefined;
|
||||||
|
|
||||||
|
// Re-embed the points using the new model
|
||||||
|
const points = records.map((record) => ({
|
||||||
|
// Keep the original ID to ensure consistency
|
||||||
|
id: record.id,
|
||||||
|
// Use the new embedding model to encode the text from the payload,
|
||||||
|
// assuming that was the original source of the embedding
|
||||||
|
vector: {
|
||||||
|
text: ((record.payload?.text as string) ?? ""),
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
// Keep the original payload
|
||||||
|
payload: record.payload,
|
||||||
|
}));
|
||||||
|
|
||||||
|
// Upsert the re-embedded points into the new collection
|
||||||
|
await client.upsert(NEW_COLLECTION, {
|
||||||
|
points,
|
||||||
|
// Only insert the point if a point with this ID does not already exist.
|
||||||
|
update_mode: "insert_only" as const,
|
||||||
|
});
|
||||||
|
|
||||||
|
// Check if we reached the end of the collection
|
||||||
|
reachedEnd = lastOffset == null;
|
||||||
|
}
|
||||||
|
// @block-end migrate-points
|
||||||
|
|
||||||
|
// @block-start search-old-collection
|
||||||
|
const results = await client.query(OLD_COLLECTION, {
|
||||||
|
query: {
|
||||||
|
text: "my query",
|
||||||
|
model: OLD_MODEL,
|
||||||
|
},
|
||||||
|
limit: 10,
|
||||||
|
});
|
||||||
|
// @block-end search-old-collection
|
||||||
|
|
||||||
|
// @block-start search-new-collection
|
||||||
|
const resultsNew = await client.query(NEW_COLLECTION, {
|
||||||
|
query: {
|
||||||
|
text: "my query",
|
||||||
|
model: NEW_MODEL,
|
||||||
|
},
|
||||||
|
limit: 10,
|
||||||
|
});
|
||||||
|
// @block-end search-new-collection
|
||||||
+9
-105
@@ -29,25 +29,14 @@ Re-embedding requires access to the original data used to create the embeddings.
|
|||||||
|
|
||||||
The solution outlined in this tutorial only works for upsert operations. If you use deletes or partial updates, it is necessary to pause those operations during the migration or implement additional logic to handle them.
|
The solution outlined in this tutorial only works for upsert operations. If you use deletes or partial updates, it is necessary to pause those operations during the migration or implement additional logic to handle them.
|
||||||
|
|
||||||
|
This tutorial assumes you use [Qdrant Cloud Inference](/documentation/concepts/inference/#qdrant-cloud-inference) to generate vector embeddings. If you manage your own embedding infrastructure, you can apply the same principles, but you will need to adapt the code examples to use your embedding service.
|
||||||
|
|
||||||
## Step 1: Create a New Collection
|
## Step 1: Create a New Collection
|
||||||
|
|
||||||
The first step is to create a new collection in Qdrant that will be used to store the new
|
The first step is to create a new collection in Qdrant that will be used to store the new
|
||||||
embeddings, compatible with the new model in terms of vector size and similarity function.
|
embeddings, compatible with the new model in terms of vector size and similarity function.
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-model-migration/" block="create-new-collection" >}}
|
||||||
from qdrant_client import QdrantClient, models
|
|
||||||
|
|
||||||
client = QdrantClient(...)
|
|
||||||
client.create_collection(
|
|
||||||
collection_name=NEW_COLLECTION,
|
|
||||||
vectors_config=(
|
|
||||||
models.VectorParams(
|
|
||||||
size=512, # Size of the new embedding vectors
|
|
||||||
distance=models.Distance.COSINE # Similarity function for the new model
|
|
||||||
)
|
|
||||||
)
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
Now is also a good moment to consider changing any other settings for the collection, like custom sharding, replication factor, etc. Switching the model may be a good opportunity to improve the performance of your search.
|
Now is also a good moment to consider changing any other settings for the collection, like custom sharding, replication factor, etc. Switching the model may be a good opportunity to improve the performance of your search.
|
||||||
|
|
||||||
@@ -60,34 +49,11 @@ To ensure that both collections are kept up-to-date during the migration, you ne
|
|||||||
|
|
||||||
Ideally, the data in Qdrant is updated by an update service reading from an update queue. This service is responsible for embedding the documents and writing them to Qdrant. It uses code similar to this:
|
Ideally, the data in Qdrant is updated by an update service reading from an update queue. This service is responsible for embedding the documents and writing them to Qdrant. It uses code similar to this:
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-model-migration/" block="upsert-old-collection" >}}
|
||||||
client.upsert(
|
|
||||||
collection_name=OLD_COLLECTION,
|
|
||||||
points=[
|
|
||||||
models.PointStruct(
|
|
||||||
id=1,
|
|
||||||
vector=encode(text="Example document", model_name=OLD_MODEL),
|
|
||||||
payload={"text": "Example document"}
|
|
||||||
)
|
|
||||||
]
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
To update the new collection, deploy a second service that updates the new collection in parallel with the existing one. This service uses the new embedding model to encode the documents and writes them to the new collection:
|
To update the new collection, deploy a second service that updates the new collection in parallel with the existing one. This service uses the new embedding model to encode the documents and writes them to the new collection:
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-model-migration/" block="upsert-new-collection" >}}
|
||||||
client.upsert(
|
|
||||||
collection_name=NEW_COLLECTION,
|
|
||||||
points=[
|
|
||||||
models.PointStruct(
|
|
||||||
id=1,
|
|
||||||
# Use the new embedding model to encode the document
|
|
||||||
vector=encode(text="Example document", model_name=NEW_MODEL),
|
|
||||||
payload={"text": "Example document"}
|
|
||||||
)
|
|
||||||
]
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
A good practice is to always ensure that both operations succeed. Any errors need to be handled on the client side. You could store errors in a log or "dead letter queue" for later processing. Transient errors can be retried at a later time. Other errors need to be analyzed and addressed accordingly.
|
A good practice is to always ensure that both operations succeed. Any errors need to be handled on the client side. You could store errors in a log or "dead letter queue" for later processing. Transient errors can be retried at a later time. Other errors need to be analyzed and addressed accordingly.
|
||||||
|
|
||||||
@@ -116,63 +82,13 @@ in parallel with the regular upsert services.
|
|||||||
|
|
||||||
The migration process reads the points from the old collection, re-embeds them using the new model, and writes them to the new collection, making sure not to overwrite existing points inserted by the update service. Here's an example of what the code for such a migration process could look like:
|
The migration process reads the points from the old collection, re-embeds them using the new model, and writes them to the new collection, making sure not to overwrite existing points inserted by the update service. Here's an example of what the code for such a migration process could look like:
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-model-migration/" block="migrate-points" >}}
|
||||||
last_offset = None
|
|
||||||
batch_size = 100 # Number of points to read in each batch
|
|
||||||
reached_end = False
|
|
||||||
|
|
||||||
while not reached_end:
|
|
||||||
# Get the next batch of points from the old collection
|
|
||||||
records, last_offset = client.scroll(
|
|
||||||
collection_name=OLD_COLLECTION,
|
|
||||||
limit=batch_size,
|
|
||||||
offset=last_offset,
|
|
||||||
# Include payloads in the response, as we need them to re-embed the vectors
|
|
||||||
with_payload=True,
|
|
||||||
# We don't need the old vectors, so let's save on the bandwidth
|
|
||||||
with_vectors=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
# Re-embed the points using the new model
|
|
||||||
upsert_operations = [
|
|
||||||
models.UpsertOperation(
|
|
||||||
upsert=models.PointsList(
|
|
||||||
points=[models.PointStruct(
|
|
||||||
# Keep the original ID to ensure consistency
|
|
||||||
id=record.id,
|
|
||||||
# Use the new embedding model to encode the text from the payload,
|
|
||||||
# assuming that was the original source of the embedding
|
|
||||||
vector=encode(record.payload.get("text"), model_name=NEW_MODEL),
|
|
||||||
# Keep the original payload
|
|
||||||
payload=record.payload
|
|
||||||
)],
|
|
||||||
# Only insert the point if a point with this ID does not already exist.
|
|
||||||
update_filter=models.Filter(
|
|
||||||
must_not=[
|
|
||||||
models.HasIdCondition(has_id=[record.id]),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
)
|
|
||||||
)
|
|
||||||
for record in records
|
|
||||||
]
|
|
||||||
|
|
||||||
# Upsert the re-embedded points into the new collection
|
|
||||||
client.batch_update_points(
|
|
||||||
collection_name=NEW_COLLECTION,
|
|
||||||
update_operations=upsert_operations
|
|
||||||
)
|
|
||||||
|
|
||||||
# Check if we reached the end of the collection
|
|
||||||
reached_end = (last_offset == None)
|
|
||||||
```
|
|
||||||
|
|
||||||
Breaking down this code step by step:
|
Breaking down this code step by step:
|
||||||
|
|
||||||
- Data is read from the old collection in batches of 100 points using a [scroll](/documentation/concepts/points/#scroll-points). The `last_offset` variable keeps track of the scroll position in the collection.
|
- Data is read from the old collection in batches of 100 points using a [scroll](/documentation/concepts/points/#scroll-points). The `last_offset` variable keeps track of the scroll position in the collection.
|
||||||
- For each batch of points, the process re-embeds the vectors using the new embedding model. It assumes that the original text used for embedding is stored in the payload under the key `text`.
|
- For each batch of points, the process re-embeds the vectors using the new embedding model. It assumes that the original text used for embedding is stored in the payload under the key `text`.
|
||||||
- With the re-embedded vectors, it prepares [conditional upsert operations](/documentation/concepts/points/#conditional-updates) for the new collection, keeping the original IDs and payloads. The conditional upserts use a filter condition to ensure that a point is only inserted if it does not already exist in the new collection. The filter checks whether a point with the given ID already exists. A point is only upserted if the ID does not exist in the new collection. This prevents overwriting newer updates from the regular update service.
|
- With the re-embedded vectors, it upserts the points into the new collection, keeping the original IDs and payloads. The upserts use [insert-only mode](/documentation/concepts/points/#update-mode) to ensure that a point is only inserted if it does not already exist in the new collection. This prevents overwriting newer updates from the regular update service.
|
||||||
- Finally, the process uses a [batch update](/documentation/concepts/points/#batch-update) to upsert the re-embedded points into the new collection. Note that it uses `batch_update_points` instead of `upsert`, because `batch_update_points` allows you to specify an update condition per upsert operation.
|
|
||||||
|
|
||||||
This kind of migration process can take some time, and the offset can be stored in a persistent way, so you can resume the migration process in case of a failure. You can use a database, a file, or any other persistent storage to keep track of the last offset. Having said that, because the conditional upserts would not overwrite any points in the new collection, you could safely restart the migration process from the beginning if needed.
|
This kind of migration process can take some time, and the offset can be stored in a persistent way, so you can resume the migration process in case of a failure. You can use a database, a file, or any other persistent storage to keep track of the last offset. Having said that, because the conditional upserts would not overwrite any points in the new collection, you could safely restart the migration process from the beginning if needed.
|
||||||
|
|
||||||
@@ -185,23 +101,11 @@ Once the migration process is complete, and all the points from the old collecti
|
|||||||
|
|
||||||
If these values are hardcoded in your application, you will need to change them directly in the code and deploy a new version of your application. For example, if your current search code looks like this:
|
If these values are hardcoded in your application, you will need to change them directly in the code and deploy a new version of your application. For example, if your current search code looks like this:
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-model-migration/" block="search-old-collection" >}}
|
||||||
results = client.query_points(
|
|
||||||
collection_name=OLD_COLLECTION,
|
|
||||||
query=encode(text="my query", model_name=OLD_MODEL), # Old query vector
|
|
||||||
limit=10,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
You need to change it in the following way:
|
You need to change it in the following way:
|
||||||
|
|
||||||
```python
|
{{< code-snippet path="/documentation/headless/snippets/tutorial-model-migration/" block="search-new-collection" >}}
|
||||||
results = client.query_points(
|
|
||||||
collection_name=NEW_COLLECTION,
|
|
||||||
query=encode(text="my query", model_name=NEW_MODEL), # New query vector
|
|
||||||
limit=10,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
## Step 5: Wrapping Up
|
## Step 5: Wrapping Up
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user