Add guide about optimizing FastEmbed throughput (#2195)

* Add guide about optimizing FastEmbed throughput

* Update weights

* Review feedback
This commit is contained in:
Abdon Pijpelink
2026-03-23 16:53:06 +01:00
committed by GitHub
parent ea0de4c12e
commit e9a691a15a
19 changed files with 252 additions and 7 deletions
@@ -0,0 +1,7 @@
```python
client = QdrantClient(
url=QDRANT_URL,
api_key=QDRANT_API_KEY,
local_inference_batch_size=256, # FastEmbed batch size
)
```
@@ -0,0 +1,13 @@
```python
point = models.PointStruct(
id=1,
vector=models.Document(
text="The text to embed",
model="BAAI/bge-small-en-v1.5",
options={
"lazy_load": True,
"cuda": True,
},
)
)
```
@@ -0,0 +1,12 @@
```python
point = models.PointStruct(
id=1,
vector=models.Document(
text="The text to embed",
model="BAAI/bge-small-en-v1.5",
options={
"lazy_load": True,
},
)
)
```
@@ -0,0 +1,38 @@
```python
from qdrant_client import QdrantClient, models
client = QdrantClient(
url=QDRANT_URL,
api_key=QDRANT_API_KEY,
local_inference_batch_size=256, # FastEmbed batch size
)
point = models.PointStruct(
id=1,
vector=models.Document(
text="The text to embed",
model="BAAI/bge-small-en-v1.5",
options={
"lazy_load": True,
},
)
)
point = models.PointStruct(
id=1,
vector=models.Document(
text="The text to embed",
model="BAAI/bge-small-en-v1.5",
options={
"lazy_load": True,
"cuda": True,
},
)
)
client.upload_points(
collection_name=COLLECTION_NAME,
points=points,
parallel=4 # use 4 workers to process documents in parallel
)
```
@@ -0,0 +1,7 @@
```python
client.upload_points(
collection_name=COLLECTION_NAME,
points=points,
parallel=4 # use 4 workers to process documents in parallel
)
```
@@ -0,0 +1,51 @@
from qdrant_client import QdrantClient, models
# @hide-start
QDRANT_URL=""
QDRANT_API_KEY=""
points: list[models.PointStruct] = []
COLLECTION_NAME=""
# @hide-end
# @block-start client-connection
client = QdrantClient(
url=QDRANT_URL,
api_key=QDRANT_API_KEY,
local_inference_batch_size=256, # FastEmbed batch size
)
# @block-end client-connection
# @block-start lazy-load
point = models.PointStruct(
id=1,
vector=models.Document(
text="The text to embed",
model="BAAI/bge-small-en-v1.5",
options={
"lazy_load": True,
},
)
)
# @block-end lazy-load
# @block-start lazy-load-gpu
point = models.PointStruct(
id=1,
vector=models.Document(
text="The text to embed",
model="BAAI/bge-small-en-v1.5",
options={
"lazy_load": True,
"cuda": True,
},
)
)
# @block-end lazy-load-gpu
# @block-start upload-data
client.upload_points(
collection_name=COLLECTION_NAME,
points=points,
parallel=4 # use 4 workers to process documents in parallel
)
# @block-end upload-data
@@ -0,0 +1,3 @@
```python
embeddings = list(model.embed(docs, batch_size=256, parallel=4))
```
@@ -0,0 +1,8 @@
```python
model = TextEmbedding(
model_name="BAAI/bge-small-en-v1.5",
lazy_load=True, # don't load the model until first embed call
cuda=True, # enable GPU acceleration
device_ids=[0, 1], # spread workers across GPUs 0 and 1
)
```
@@ -0,0 +1,6 @@
```python
model = TextEmbedding(
model_name="BAAI/bge-small-en-v1.5",
lazy_load=True, # don't load the model until first embed call
)
```
@@ -0,0 +1,17 @@
```python
from fastembed import TextEmbedding
model = TextEmbedding(
model_name="BAAI/bge-small-en-v1.5",
lazy_load=True, # don't load the model until first embed call
)
model = TextEmbedding(
model_name="BAAI/bge-small-en-v1.5",
lazy_load=True, # don't load the model until first embed call
cuda=True, # enable GPU acceleration
device_ids=[0, 1], # spread workers across GPUs 0 and 1
)
embeddings = list(model.embed(docs, batch_size=256, parallel=4))
```
@@ -0,0 +1,25 @@
from fastembed import TextEmbedding
# @block-start lazy-load
model = TextEmbedding(
model_name="BAAI/bge-small-en-v1.5",
lazy_load=True, # don't load the model until first embed call
)
# @block-end lazy-load
# @block-start lazy-load-gpu
model = TextEmbedding(
model_name="BAAI/bge-small-en-v1.5",
lazy_load=True, # don't load the model until first embed call
cuda=True, # enable GPU acceleration
device_ids=[0, 1], # spread workers across GPUs 0 and 1
)
# @block-end lazy-load-gpu
# @hide-start
docs = ["", "", ""]
# @hide-end
# @block-start embed
embeddings = list(model.embed(docs, batch_size=256, parallel=4))
# @block-end embed