mirror of
https://github.com/qdrant/landing_page.git
synced 2026-10-04 02:18:29 +02:00
Add guide about optimizing FastEmbed throughput (#2195)
* Add guide about optimizing FastEmbed throughput * Update weights * Review feedback
This commit is contained in:
+7
@@ -0,0 +1,7 @@
|
||||
```python
|
||||
client = QdrantClient(
|
||||
url=QDRANT_URL,
|
||||
api_key=QDRANT_API_KEY,
|
||||
local_inference_batch_size=256, # FastEmbed batch size
|
||||
)
|
||||
```
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
```python
|
||||
point = models.PointStruct(
|
||||
id=1,
|
||||
vector=models.Document(
|
||||
text="The text to embed",
|
||||
model="BAAI/bge-small-en-v1.5",
|
||||
options={
|
||||
"lazy_load": True,
|
||||
"cuda": True,
|
||||
},
|
||||
)
|
||||
)
|
||||
```
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
```python
|
||||
point = models.PointStruct(
|
||||
id=1,
|
||||
vector=models.Document(
|
||||
text="The text to embed",
|
||||
model="BAAI/bge-small-en-v1.5",
|
||||
options={
|
||||
"lazy_load": True,
|
||||
},
|
||||
)
|
||||
)
|
||||
```
|
||||
+38
@@ -0,0 +1,38 @@
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient(
|
||||
url=QDRANT_URL,
|
||||
api_key=QDRANT_API_KEY,
|
||||
local_inference_batch_size=256, # FastEmbed batch size
|
||||
)
|
||||
|
||||
point = models.PointStruct(
|
||||
id=1,
|
||||
vector=models.Document(
|
||||
text="The text to embed",
|
||||
model="BAAI/bge-small-en-v1.5",
|
||||
options={
|
||||
"lazy_load": True,
|
||||
},
|
||||
)
|
||||
)
|
||||
|
||||
point = models.PointStruct(
|
||||
id=1,
|
||||
vector=models.Document(
|
||||
text="The text to embed",
|
||||
model="BAAI/bge-small-en-v1.5",
|
||||
options={
|
||||
"lazy_load": True,
|
||||
"cuda": True,
|
||||
},
|
||||
)
|
||||
)
|
||||
|
||||
client.upload_points(
|
||||
collection_name=COLLECTION_NAME,
|
||||
points=points,
|
||||
parallel=4 # use 4 workers to process documents in parallel
|
||||
)
|
||||
```
|
||||
+7
@@ -0,0 +1,7 @@
|
||||
```python
|
||||
client.upload_points(
|
||||
collection_name=COLLECTION_NAME,
|
||||
points=points,
|
||||
parallel=4 # use 4 workers to process documents in parallel
|
||||
)
|
||||
```
|
||||
+51
@@ -0,0 +1,51 @@
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
# @hide-start
|
||||
QDRANT_URL=""
|
||||
QDRANT_API_KEY=""
|
||||
points: list[models.PointStruct] = []
|
||||
COLLECTION_NAME=""
|
||||
# @hide-end
|
||||
|
||||
# @block-start client-connection
|
||||
client = QdrantClient(
|
||||
url=QDRANT_URL,
|
||||
api_key=QDRANT_API_KEY,
|
||||
local_inference_batch_size=256, # FastEmbed batch size
|
||||
)
|
||||
# @block-end client-connection
|
||||
|
||||
# @block-start lazy-load
|
||||
point = models.PointStruct(
|
||||
id=1,
|
||||
vector=models.Document(
|
||||
text="The text to embed",
|
||||
model="BAAI/bge-small-en-v1.5",
|
||||
options={
|
||||
"lazy_load": True,
|
||||
},
|
||||
)
|
||||
)
|
||||
# @block-end lazy-load
|
||||
|
||||
# @block-start lazy-load-gpu
|
||||
point = models.PointStruct(
|
||||
id=1,
|
||||
vector=models.Document(
|
||||
text="The text to embed",
|
||||
model="BAAI/bge-small-en-v1.5",
|
||||
options={
|
||||
"lazy_load": True,
|
||||
"cuda": True,
|
||||
},
|
||||
)
|
||||
)
|
||||
# @block-end lazy-load-gpu
|
||||
|
||||
# @block-start upload-data
|
||||
client.upload_points(
|
||||
collection_name=COLLECTION_NAME,
|
||||
points=points,
|
||||
parallel=4 # use 4 workers to process documents in parallel
|
||||
)
|
||||
# @block-end upload-data
|
||||
+3
@@ -0,0 +1,3 @@
|
||||
```python
|
||||
embeddings = list(model.embed(docs, batch_size=256, parallel=4))
|
||||
```
|
||||
+8
@@ -0,0 +1,8 @@
|
||||
```python
|
||||
model = TextEmbedding(
|
||||
model_name="BAAI/bge-small-en-v1.5",
|
||||
lazy_load=True, # don't load the model until first embed call
|
||||
cuda=True, # enable GPU acceleration
|
||||
device_ids=[0, 1], # spread workers across GPUs 0 and 1
|
||||
)
|
||||
```
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
```python
|
||||
model = TextEmbedding(
|
||||
model_name="BAAI/bge-small-en-v1.5",
|
||||
lazy_load=True, # don't load the model until first embed call
|
||||
)
|
||||
```
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
```python
|
||||
from fastembed import TextEmbedding
|
||||
|
||||
model = TextEmbedding(
|
||||
model_name="BAAI/bge-small-en-v1.5",
|
||||
lazy_load=True, # don't load the model until first embed call
|
||||
)
|
||||
|
||||
model = TextEmbedding(
|
||||
model_name="BAAI/bge-small-en-v1.5",
|
||||
lazy_load=True, # don't load the model until first embed call
|
||||
cuda=True, # enable GPU acceleration
|
||||
device_ids=[0, 1], # spread workers across GPUs 0 and 1
|
||||
)
|
||||
|
||||
embeddings = list(model.embed(docs, batch_size=256, parallel=4))
|
||||
```
|
||||
+25
@@ -0,0 +1,25 @@
|
||||
from fastembed import TextEmbedding
|
||||
|
||||
# @block-start lazy-load
|
||||
model = TextEmbedding(
|
||||
model_name="BAAI/bge-small-en-v1.5",
|
||||
lazy_load=True, # don't load the model until first embed call
|
||||
)
|
||||
# @block-end lazy-load
|
||||
|
||||
# @block-start lazy-load-gpu
|
||||
model = TextEmbedding(
|
||||
model_name="BAAI/bge-small-en-v1.5",
|
||||
lazy_load=True, # don't load the model until first embed call
|
||||
cuda=True, # enable GPU acceleration
|
||||
device_ids=[0, 1], # spread workers across GPUs 0 and 1
|
||||
)
|
||||
# @block-end lazy-load-gpu
|
||||
|
||||
# @hide-start
|
||||
docs = ["", "", ""]
|
||||
# @hide-end
|
||||
|
||||
# @block-start embed
|
||||
embeddings = list(model.embed(docs, batch_size=256, parallel=4))
|
||||
# @block-end embed
|
||||
Reference in New Issue
Block a user