Merge branch 'hybrid-cloud-dev' into hybrid-cloud/tutorial/stackit-aleph-alpha-contract-management
@@ -274,3 +274,48 @@ From the root of the project:
|
||||
```bash
|
||||
sass --watch --style=compressed ./qdrant-landing/themes/qdrant/static/css/pages/marketing-landing.scss ./qdrant-landing/themes/qdrant/static/css/marketing-landing.css
|
||||
```
|
||||
|
||||
## SEO
|
||||
|
||||
### Structured data (Schema.org, JSON-LD)
|
||||
|
||||
Structured data is a standardized format for providing information about a page and classifying the page content. It is used by search engines to understand the content of the page and to display rich snippets in search results.
|
||||
|
||||
We use JSON-LD format for structured data. Data is stored in JSON files in the `/assets/schema` directory. If no specific schema is provided for a page, the default schema is used based on the page type as defined in the `qdrant-landing/themes/qdrant/layouts/partials/seo_schema.html` file.
|
||||
|
||||
To add specific schema to a specific page, use the `seo_schema` or `seo_schema_json` parameter in the front matter of content markdown files (directory `content`).
|
||||
|
||||
To add json directly to the page, use the `seo_schema` parameter. The value should be a JSON object.
|
||||
|
||||
Example:
|
||||
|
||||
```yaml
|
||||
seo_schema: {
|
||||
"@context": "https://schema.org",
|
||||
"@type": "Organization",
|
||||
"name": "Qdrant",
|
||||
"url": "https://qdrant.io",
|
||||
"logo": "https://qdrant.io/images/logo.png",
|
||||
"sameAs": [
|
||||
"https://www.linkedin.com/company/qdrant",
|
||||
"https://twitter.com/qdrant"
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
To add a path to a JSON files with schema data, use the `seo_schema_json` parameter. This parameter should contain a list of paths to JSON files.
|
||||
The path should be relative to the `qdrant-landing/assets` directory.
|
||||
|
||||
Example:
|
||||
|
||||
```yaml
|
||||
seo_schema_json:
|
||||
- schema/schema-organization.json
|
||||
- schema/product-schema.json
|
||||
```
|
||||
|
||||
If you want to add a new schema, create a new JSON file in the `qdrant-landing/assets/schema` directory and add the path to the `seo_schema_json` parameter.
|
||||
|
||||
When use `seo_schema` and `seo_schema_json` together, `seo_schema` will be used additionally to `seo_schema_json` adding the second <script> tag with the `seo_schema` value.
|
||||
|
||||
Use `seo_schema_json` if you want to reuse the same schema for multiple pages to avoid duplication and make it easier to maintain.
|
||||
@@ -0,0 +1,24 @@
|
||||
{
|
||||
"@type": "Article",
|
||||
"@id": "{{- .Permalink -}}#article",
|
||||
"name": "{{- .Params.title | htmlEscape -}}",
|
||||
"headline": "{{- .Params.title | htmlEscape -}}",
|
||||
"image": [
|
||||
"{{- if .Params.social_preview_image -}}{{- .Params.social_preview_image | absURL -}}{{- end -}}"
|
||||
],
|
||||
"url": "{{- .Permalink -}}",
|
||||
"description": {{ $description := printf "%s" (.Params.description | plainify | replaceRE "(\n)" "" | replaceRE "[^\\w\\s:\\[\\]{}\"]" "" | htmlEscape ) -}}"{{- $description -}}",
|
||||
"abstract": {{- $abstract := .Summary -}}{{- if .Params.description -}}{{- $abstract = .Params.description -}}{{- end -}}{{ $processedAbstract := printf "%s" ($abstract | plainify | replaceRE "(\n)" "" | replaceRE "[^\\w\\s:\\[\\]{}\"]" "" | htmlEscape ) -}}"{{- $processedAbstract -}}",
|
||||
"wordCount": "{{- .WordCount -}}",
|
||||
"datePublished": "{{ .Date }}" ,
|
||||
"dateModified": "{{ .Date }}",{{- if .Params.author -}}
|
||||
"author": {
|
||||
"@type": "Person",
|
||||
"name": "{{- .Params.author | htmlEscape -}}"
|
||||
}{{- else -}}
|
||||
"author": {
|
||||
"@type": "Organization",
|
||||
"name": "Qdrant",
|
||||
"url": "https://qdrant.tech/"
|
||||
}{{- end -}}
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
{
|
||||
"@type": "Organization",
|
||||
"@id": "{{- .Permalink -}}#organization",
|
||||
"name": "Qdrant",
|
||||
"legalName" : "Qdrant Solutions GmbH",
|
||||
"url": "https://qdrant.tech",
|
||||
"email": "info@qdrant.com",
|
||||
"logo": "https://qdrant.tech/images/logo_with_text.png",
|
||||
"description" : "{{- .Site.Params.description | htmlEscape -}}",
|
||||
"keywords" : [ {{- if .Params.Keywords -}}{{- range .Params.Keywords -}}"{{ . }}", {{- end -}}{{- else if .Site.Params.Keywords -}}{{- range .Site.Params.Keywords -}}"{{ . }}", {{- end -}}{{- end -}} "Qdrant" ],
|
||||
"foundingDate": "2021",
|
||||
"founders": [
|
||||
{
|
||||
"@type": "Person",
|
||||
"name": "{{ .Site.Params.Author }}"
|
||||
}, {
|
||||
"@type": "Person",
|
||||
"name": "Andre Zayarni"
|
||||
}
|
||||
],
|
||||
"location": "Berlin, Germany",
|
||||
"address": {
|
||||
"@type": "PostalAddress",
|
||||
"streetAddress": "Chausseestraße 86",
|
||||
"addressLocality": "Berlin",
|
||||
"addressRegion": "Berlin",
|
||||
"postalCode": "10115",
|
||||
"addressCountry": "DE"
|
||||
},
|
||||
"contactPoint": {
|
||||
"@type": "ContactPoint",
|
||||
"contactType": "customer support",
|
||||
"telephone": "+49 3040797694",
|
||||
"email": "info@qdrant.com"
|
||||
},
|
||||
"sameAs": [
|
||||
"{{ .Site.Params.github }}",
|
||||
"{{ .Site.Params.discord }}",
|
||||
"{{ .Site.Params.youtube }}",
|
||||
"https://www.linkedin.com/company/qdrant/",
|
||||
"https://twitter.com/qdrant_engine"
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
{
|
||||
"@type": "Product",
|
||||
"@id": "{{- .Permalink -}}#product",
|
||||
"brand": {
|
||||
"@id": "https://qdrant.tech",
|
||||
"@type": "Organization",
|
||||
"name": "Qdrant Vector Database"
|
||||
},
|
||||
"description" : "{{- .Site.Params.description | htmlEscape -}}",
|
||||
"keywords" : [ {{- if .Params.Keywords -}}{{- range .Params.Keywords -}}"{{ . }}", {{- end -}}{{- else if .Site.Params.Keywords -}}{{- range .Site.Params.Keywords -}}"{{ . }}", {{- end -}}{{- end -}} "Qdrant" ],
|
||||
"name": "Qdrant",
|
||||
"image": "{{- if .Params.social_preview_image -}}{{- .Params.social_preview_image | absURL -}}{{- end -}}",
|
||||
"offers": {
|
||||
"@type": "AggregateOffer",
|
||||
"lowPrice": "0",
|
||||
"offerCount": "3",
|
||||
"priceCurrency": "USD",
|
||||
"offers": [
|
||||
{
|
||||
"@type": "Offer",
|
||||
"priceSpecification": {
|
||||
"@type": "PriceSpecification",
|
||||
"price": "free",
|
||||
"priceCurrency": "USD",
|
||||
"name": "Community",
|
||||
"url": "https://qdrant.tech/documentation/quick-start/"
|
||||
}
|
||||
},
|
||||
{
|
||||
"@type": "Offer",
|
||||
"priceSpecification": {
|
||||
"@type": "PriceSpecification",
|
||||
"price": "from 25$",
|
||||
"priceCurrency": "USD",
|
||||
"name": "Managed Cloud",
|
||||
"url": "https://qdrant.to/cloud"
|
||||
}
|
||||
},
|
||||
{
|
||||
"@type": "Offer",
|
||||
"priceSpecification": {
|
||||
"@type": "PriceSpecification",
|
||||
"price": "on request",
|
||||
"priceCurrency": "USD",
|
||||
"name": "Enterprise",
|
||||
"url": "https://qdrant.to/contact-us"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -54,7 +54,7 @@ As embeddings are vectors, one can apply a simple function to calculate the simi
|
||||
So with similarity learning, all we need to do is provide pairs of correct questions and answers.
|
||||
And then, the model will learn to distinguish proper answers by the similarity of embeddings.
|
||||
|
||||
>If you want to learn more about similarity learning and applications, check out this [article](https://blog.qdrant.tech/neural-search-tutorial-3f034ab13adc) which might be an asset.
|
||||
>If you want to learn more about similarity learning and applications, check out this [article](https://qdrant.tech/documentation/tutorials/neural-search/) which might be an asset.
|
||||
|
||||
## Let's build
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ set up your collections.
|
||||
|
||||
Previously, you had to send multiple requests to the Qdrant API to perform multiple non-related tasks. However, this
|
||||
can cause significant network overhead and slow down the process, especially if you have a poor connection speed.
|
||||
Fortunately, the [new batch search feature](https://blog.qdrant.tech/batch-vector-search-with-qdrant-8c4d598179d5) allows
|
||||
Fortunately, the [new batch search feature](https://qdrant.tech/documentation/concepts/search/#batch-search-api) allows
|
||||
you to avoid this issue. With just one API call, Qdrant will handle multiple search requests in the most efficient way
|
||||
possible. This means that you can perform multiple tasks simultaneously without having to worry about network overhead
|
||||
or slow performance.
|
||||
@@ -37,14 +37,13 @@ or slow performance.
|
||||
To make our application accessible to ARM users, we have compiled it specifically for that platform. If it is not
|
||||
compiled for ARM, the device will have to emulate it, which can slow down performance. To ensure the best possible
|
||||
experience for ARM users, we have created Docker images specifically for that platform. Keep in mind that using
|
||||
a limited set of processor instructions may affect the performance of your vector search. Therefore, [we have tested
|
||||
both ARM and non-ARM architectures using similar setups to understand the potential impact on performance
|
||||
](https://blog.qdrant.tech/qdrant-supports-arm-architecture-363e92aa5026).
|
||||
a limited set of processor instructions may affect the performance of your vector search. Therefore, we have tested
|
||||
both ARM and non-ARM architectures using similar setups to understand the potential impact on performance.
|
||||
|
||||
## Full-text filtering
|
||||
|
||||
Qdrant is a vector database that allows you to quickly search for the nearest neighbors. However, you may need to apply
|
||||
additional filters on top of the semantic search. Up until version 0.10, Qdrant only supported keyword filters. With the
|
||||
release of Qdrant 0.10, [you can now use full-text filters](https://blog.qdrant.tech/qdrant-introduces-full-text-filters-and-indexes-9a032fcb5fa)
|
||||
release of Qdrant 0.10, [you can now use full-text filters](https://qdrant.tech/documentation/concepts/filtering/#full-text-match)
|
||||
as well. This new filter type can be used on its own or in combination with other filter types to provide even more
|
||||
flexibility in your searches.
|
||||
|
||||
@@ -175,6 +175,6 @@ After building your RAG chatbot, you'll be able to evaluate its performance agai
|
||||
|
||||
## What’s next?
|
||||
|
||||
Have a RAG project you want to bring to life? Join our [Discord community](discord.gg/qdrant) where we’re always sharing tips and answering questions on vector search and retrieval.
|
||||
Have a RAG project you want to bring to life? Join our [Discord community](https://discord.gg/qdrant) where we’re always sharing tips and answering questions on vector search and retrieval.
|
||||
|
||||
Learn more about how to properly evaluate your RAG responses: [Evaluating Retrieval Augmented Generation - a framework for assessment](https://superlinked.com/vectorhub/evaluating-retrieval-augmented-generation-a-framework-for-assessment).
|
||||
@@ -0,0 +1,65 @@
|
||||
---
|
||||
title: "Response to CVE-2024-2221: Arbitrary file upload vulnerability"
|
||||
draft: false
|
||||
slug: cve-2024-2221-response
|
||||
short_description: Qdrant keeps your systems secure
|
||||
description: Upgrade your deployments to at least v1.8.0. Cloud deployments not materially affected.
|
||||
preview_image: /blog/cve-2024-2221/cve-2024-2221-response-social-preview.png
|
||||
|
||||
# social_preview_image: /blog/Article-Image.png # Optional image used for link previews
|
||||
# title_preview_image: /blog/Article-Image.png # Optional image used for blog post title
|
||||
# small_preview_image: /blog/Article-Image.png # Optional image used for small preview in the list of blog posts
|
||||
date: 2024-04-05T13:00:00-07:00
|
||||
author: Mike Jang
|
||||
featured: false
|
||||
tags:
|
||||
- cve
|
||||
- security
|
||||
weight: 0 # Change this weight to change order of posts
|
||||
# For more guidance, see https://github.com/qdrant/landing_page?tab=readme-ov-file#blog
|
||||
---
|
||||
|
||||
### Summary
|
||||
|
||||
A security vulnerability has been discovered in Qdrant affecting all versions
|
||||
prior to v1.8, described in [CVE-2024-2221](https://cve.mitre.org/cgi-bin/cvename.cgi?name=CVE-2024-2221).
|
||||
The vulnerability allows an attacker to upload arbitrary files to the
|
||||
filesystem, which can be used to gain remote code execution.
|
||||
|
||||
The vulnerability does not materially affect Qdrant cloud deployments, as that
|
||||
filesystem is read-only and authentication is enabled by default. At worst,
|
||||
the vulnerability could be used by an authenticated user to crash a cluster,
|
||||
which is already possible, such as by uploading more vectors than can fit in RAM.
|
||||
|
||||
Qdrant has addressed the vulnerability in v1.8.3 and above with code that
|
||||
restricts file uploads to a folder dedicated to that purpose.
|
||||
|
||||
### Action
|
||||
|
||||
Check the current version of your Qdrant deployment. Upgrade if your deployment
|
||||
is not at least v1.8.3.
|
||||
|
||||
To confirm the version of your Qdrant deployment in the cloud or on your local
|
||||
or cloud system, run an API GET call, as described in the [Qdrant Quickstart
|
||||
guide](https://qdrant.tech/documentation/cloud/quickstart-cloud/#step-2-test-cluster-access).
|
||||
If your Qdrant deployment is local, you do not need an API key.
|
||||
|
||||
Your next step depends on how you installed Qdrant. For details, read the
|
||||
[Qdrant Installation](https://qdrant.tech/documentation/guides/installation/)
|
||||
guide.
|
||||
|
||||
#### If you use the Qdrant container or binary
|
||||
|
||||
Upgrade your deployment. Run the commands in the applicable section of the
|
||||
[Qdrant Installation](https://qdrant.tech/documentation/guides/installation/)
|
||||
guide. The default commands automatically pull the latest version of Qdrant.
|
||||
|
||||
#### If you use the Qdrant helm chart
|
||||
|
||||
If you’ve set up Qdrant on kubernetes using a helm chart, follow the README in
|
||||
the [qdrant-helm](https://github.com/qdrant/qdrant-helm/tree/main?tab=readme-ov-file#upgrading) repository.
|
||||
Make sure applicable configuration files point to version v1.8.3 or above.
|
||||
|
||||
#### If you use the Qdrant cloud
|
||||
|
||||
No action is required. This vulnerability does not materially affect you. However, we suggest that you upgrade your cloud deployment to the latest version.
|
||||
@@ -0,0 +1,59 @@
|
||||
---
|
||||
draft: false
|
||||
title: "Introducing FastLLM: Qdrant’s Revolutionary LLM"
|
||||
short_description: The most powerful LLM known to human...or LLM.
|
||||
description: Lightweight and open-source. Custom made for RAG and completely integrated with Qdrant.
|
||||
preview_image: /blog/fastllm-announcement/fastllm.png
|
||||
date: 2024-04-01T00:00:00Z
|
||||
author: David Myriel
|
||||
featured: false
|
||||
weight: 0
|
||||
tags:
|
||||
- Qdrant
|
||||
- FastEmbed
|
||||
- LLM
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
Today, we're happy to announce that **FastLLM (FLLM)**, our lightweight Language Model tailored specifically for Retrieval Augmented Generation (RAG) use cases, has officially entered Early Access!
|
||||
|
||||
Developed to seamlessly integrate with Qdrant, **FastLLM** represents a significant leap forward in AI-driven content generation. Up to this point, LLM’s could only handle up to a few million tokens.
|
||||
|
||||
**As of today, FLLM offers a context window of 1 billion tokens.**
|
||||
|
||||
However, what sets FastLLM apart is its optimized architecture, making it the ideal choice for RAG applications. With minimal effort, you can combine FastLLM and Qdrant to launch applications that process vast amounts of data. Leveraging the power of Qdrant's scalability features, FastLLM promises to revolutionize how enterprise AI applications generate and retrieve content at massive scale.
|
||||
|
||||
> *“First we introduced [FastEmbed](https://github.com/qdrant/fastembed). But then we thought - why stop there? Embedding is useful and all, but our users should do everything from within the Qdrant ecosystem. FastLLM is just the natural progression towards a large-scale consolidation of AI tools.” Andre Zayarni, President & CEO, Qdrant*
|
||||
>
|
||||
|
||||
## Going Big: Quality & Quantity
|
||||
|
||||
Very soon, an LLM will come out with a context window so wide, it will completely eliminate any value a measly vector database can add.
|
||||
|
||||
***We know this. That’s why we trained our own LLM to obliterate the competition. Also, in case vector databases go under, at least we'll have an LLM left!***
|
||||
|
||||
As soon as we entered Series A, we knew it was time to ramp up our training efforts. FLLM was trained on 300,000 NVIDIA H100s connected by 5Tbps Infiniband. It took weeks to fully train the model, but our unified efforts produced the most powerful LLM known to human…..or LLM.
|
||||
|
||||
We don’t see how any other company can compete with FastLLM. Most of our competitors will soon be burning through graphics cards trying to get to the next best thing. But it is too late. By this time next year, we will have left them in the dust.
|
||||
|
||||
> ***“Everyone has an LLM, so why shouldn’t we? Let’s face it - the more products and features you offer, the more they will sign up. Sure, this is a major pivot…but life is all about being bold.”*** *David Myriel, Director of Product Education, Qdrant*
|
||||
>
|
||||
|
||||
## Extreme Performance
|
||||
|
||||
Qdrant’s R&D is proud to stand behind the most dramatic benchmark results. Across a range of standard benchmarks, FLLM surpasses every single model in existence. In the [Needle In A Haystack](https://github.com/gkamradt/LLMTest_NeedleInAHaystack) (NIAH) test, FLLM found the embedded text with 100% accuracy, always within blocks containing 1 billion tokens. We actually believe FLLM can handle more than a trillion tokens, but it’s quite possible that it is hiding its true capabilities.
|
||||
|
||||
FastLLM has a fine-grained mixture-of-experts architecture and a whopping 1 trillion total parameters. As developers and researchers delve into the possibilities unlocked by this new model, they will uncover new applications, refine existing solutions, and perhaps even stumble upon unforeseen breakthroughs. As of now, we're not exactly sure what problem FLLM is solving, but hey, it's got a lot of parameters!
|
||||
|
||||
> *Our customers ask us “What can I do with an LLM this extreme?” I don’t know, but it can’t hurt to build another RAG chatbot.” Kacper Lukawski, Senior Developer Advocate, Qdrant*
|
||||
>
|
||||
|
||||
## Get Started!
|
||||
|
||||
Don't miss out on this opportunity to be at the forefront of AI innovation. Join FastLLM's Early Access program now and embark on a journey towards AI-powered excellence!
|
||||
|
||||
Stay tuned for more updates and exciting developments as we continue to push the boundaries of what's possible with AI-driven content generation.
|
||||
|
||||
Happy Generating! 🚀
|
||||
|
||||
[Sign Up for Early Access](https://qdrant.to/cloud)
|
||||
@@ -0,0 +1,185 @@
|
||||
---
|
||||
draft: true
|
||||
title: Gen AI and Vector Search - Iveta Lohovska | Vector Space Talks
|
||||
slug: gen-ai-and-vector-search
|
||||
short_description: Iveta emphasizes the importance of trustworthy AI,
|
||||
particularly when implementing it within high-stakes enterprises like
|
||||
governments and security agencies
|
||||
description: Iveta Lohovska discusses the importance of explainability and
|
||||
transparency, discussing high-stakes use cases in sectors like cybersecurity
|
||||
and climate data, and emphasizing the necessity for on-prem solutions and
|
||||
traceable vector databases to ensure data integrity and confidentiality.
|
||||
preview_image: /blog/from_cms/iveta-lohovska-bp-cropped.png
|
||||
date: 2024-04-04T21:28:00.000Z
|
||||
author: Demetrios Brinkmann
|
||||
featured: false
|
||||
tags:
|
||||
- Vector Space Talks
|
||||
- Vector Search
|
||||
- Retrieval Augmented Generation
|
||||
- GenAI
|
||||
---
|
||||
> *"In the generative AI context of AI, all foundational models have been trained on some foundational data sets that are distributed in different ways. Some are very conversational, some are very technical, some are on, let's say very strict taxonomy like healthcare or chemical structures. We call them modalities, and they have different representations.”*\
|
||||
— Iveta Lohovska
|
||||
>
|
||||
|
||||
Iveta Lohovska serves as the Chief Technologist and Principal Data Scientist for AI and Supercomputing at Hewlett Packard Enterprise (HPE), where she champions the democratization of decision intelligence and the development of ethical AI solutions. An industry leader, her multifaceted expertise encompasses natural language processing, computer vision, and data mining. Committed to leveraging technology for societal benefit, Iveta is a distinguished technical advisor to the United Nations' AI for Good program and a Data Science lecturer at the Vienna University of Applied Sciences. Her career also includes impactful roles with the World Bank Group, focusing on open data initiatives and Sustainable Development Goals (SDGs), as well as collaborations with USAID and the Gates Foundation.
|
||||
|
||||
***Listen to the episode on [Spotify](https://open.spotify.com/episode/7f1RDwp5l2Ps9N7gKubl8S?si=kCSX4HGCR12-5emokZbRfw), Apple Podcast, Podcast addicts, Castbox. You can also watch this episode on [YouTube](https://youtu.be/RsRAUO-fNaA).***
|
||||
|
||||
<iframe width="560" height="315" src="https://www.youtube.com/embed/RsRAUO-fNaA?si=s3k_-DP1U0rkPlEV" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
|
||||
|
||||
<iframe src="https://podcasters.spotify.com/pod/show/qdrant-vector-space-talk/embed/episodes/Gen-AI-and-Vector-Search---Iveta-Lohovska--Vector-Space-Talks-020-e2hnie2/a-ab48uha" height="102px" width="400px" frameborder="0" scrolling="no"></iframe>
|
||||
|
||||
## **Top takeaways:**
|
||||
|
||||
In our continuous pursuit of knowledge and understanding, especially in the evolving landscape of AI and the vector space, we brought another great Vector Space Talk episode featuring Iveta Lohovska as she talks about generative AI and vector search.
|
||||
|
||||
Iveta brings valuable insights from her work with the World Bank and as Chief Technologist at HPE, explaining the ins and outs of ethical AI implementation.
|
||||
|
||||
Here are the episode highlights:
|
||||
- Exploring the critical role of trustworthiness and explainability in AI, especially within high confidentiality use cases like government and security agencies.
|
||||
- Discussing the importance of transparency in AI models and how it impacts the handling of data and understanding the foundational datasets for vector search.
|
||||
- Iveta shares her experiences implementing generative AI in high-stakes environments, including the energy sector and policy-making, emphasizing accuracy and source credibility.
|
||||
- Strategies for managing data privacy in high-stakes sectors, the superiority of on-premises solutions for control, and the implications of opting for cloud or hybrid infrastructure.
|
||||
- Iveta's take on the maturity levels of generative AI, the ongoing development of smaller, more focused models, and the evolving landscape of AI model licensing and open-source contributions.
|
||||
|
||||
> Fun Fact: The climate agent solution showcased by Iveta helps individuals benchmark their carbon footprint and assists policymakers in drafting policy recommendations based on scientifically accurate data.
|
||||
>
|
||||
|
||||
## Show notes:
|
||||
|
||||
00:00 AI's vulnerabilities and ethical implications in practice.\
|
||||
06:28 Trust reliable sources for accurate climate data.\
|
||||
09:14 Vector database offers control and explainability.\
|
||||
13:21 On-prem vital for security and control.\
|
||||
16:47 Gen AI chat models at basic maturity.\
|
||||
19:28 Mature technical community, but slow enterprise adoption.\
|
||||
23:34 Advocates for open source but highlights complexities.\
|
||||
25:38 Unreliable information, triangle of necessities, vector space.
|
||||
|
||||
## More Quotes from Iveta:
|
||||
|
||||
*"What we have to ensure here is that every citation and every answer and augmentation by the generative AI on top of that is linked to the exact source of paper or publication, where it's coming from, to ensure that we can trace it back to where the climate information is coming from.”*\
|
||||
— Iveta Lohovska
|
||||
|
||||
*"Explainability means if you receive a certain answer based on your prompt, you can trace it back to the exact source where the embedding has been stored or the source of where the information is coming from and things.”*\
|
||||
— Iveta Lohovska
|
||||
|
||||
*"Chat GPT for conversational purposes and individual help is something very cool but when this needs to be translated into actual business use cases scenario with all the constraint of the enterprise architecture, with the constraint of the use cases, the reality changes quite dramatically.”*\
|
||||
— Iveta Lohovska
|
||||
|
||||
## Transcript:
|
||||
Demetrios:
|
||||
Look at that. We are back for another vector space talks. I'm very excited to be doing this today with you all. I am joined by none other than Sabrina again. Where are you at, Sabrina? How's it going?
|
||||
|
||||
Sabrina Aquino:
|
||||
Hey there, Demetrios. Amazing. Another episode and I'm super excited for this one. How are you doing?
|
||||
|
||||
Demetrios:
|
||||
I'm great. And we're going to bring out our guest of honor today. We are going to be talking a lot about trustworthy AI because Iveta has a background working with the World bank and focusing on the open data with that. But currently she is chief technologist and principal data scientist at HPE. And we were talking before we hit record before we went live. And we've got some hot takes that are coming up. So I'm going to bring Iveta to the stage. Where are you? There you are, our guest of honor.
|
||||
|
||||
Demetrios:
|
||||
How you doing?
|
||||
|
||||
Iveta Lohovska:
|
||||
Good. I hope you can hear me well.
|
||||
|
||||
Demetrios:
|
||||
Loud and clear. Yes.
|
||||
|
||||
Iveta Lohovska:
|
||||
Happy to join here from Vienna and thank you for the invite.
|
||||
|
||||
Demetrios:
|
||||
Yes. So I'm very excited to talk with you today. I think it's probably worth getting the TLDR on your story and why you're so passionate about trustworthiness and explainability.
|
||||
|
||||
Iveta Lohovska:
|
||||
Well, I think especially in the genaid context where if there any vulnerabilities around the solution or the training data set or any underlying context, either in the enterprise or in a smaller scale, it's just the scale that AI engine AI can achieve if it has any vulnerabilities or any weaknesses when it comes to explainability or trustworthiness or bias, it just goes explain nature. So it is to be considered and taken with high attention when it comes to those use cases. And most of my work is within an enterprise with high confidentiality use cases. So it plays a big role more than actually people will think it's on a high level. It just sounds like AI ethical principles or high level words that are very difficult to implement in technical terms. But in reality, when you hit the ground, when you hit the projects, when you work with in the context of, let's say, governments or organizations that deal with atomic energy, I see it in Vienna, the atomic agency is a neighboring one, or security agencies. Then you see the importance and the impact of those terms and the technical implications behind that.
|
||||
|
||||
Sabrina Aquino:
|
||||
That's amazing. And can you talk a little bit more about the importance of the transparency of these models and what can happen if we don't know exactly what kind of data they are being trained on?
|
||||
|
||||
Iveta Lohovska:
|
||||
I mean, this is especially relevant under our context of vector databases and vector search. Because in the generative AI context of AI, all foundational models have been trained on some foundational data sets that are distributed in different ways. Some are very conversational, some are very technical, some are on, let's say very strict taxonomy like healthcare or chemical structures. We call them modalities, and they have different representations. So, so when it comes to implementing vector search or vector database and knowing the distribution of the foundational data sets, you have better control if you introduce additional layers or additional components to have the control in your hands of where the information is coming from, where it's stored, what are the embeddings. So that helps, but it is actually quite important that you know what the foundational data sets are, so that you can predict any kind of weaknesses or vulnerabilities or penetrations that the solution or the use case of the model will face when it lands at the end user. Because we know with generative AI that is unpredictable, we know we can implement guardrails. They're already solutions.
|
||||
|
||||
Iveta Lohovska:
|
||||
We know they're not 100, they don't give you 100% certainty, but they are definitely use cases and work where you need to hit the hundred percent certainty, especially intelligence, cybersecurity and healthcare.
|
||||
|
||||
Demetrios:
|
||||
Yeah, that's something that I wanted to dig into a little bit. More of these high stakes use cases feel like you can't. I don't know. I talk with a lot of people about at this current time, it's very risky to try and use specifically generative AI for those high stakes use cases. Have you seen people that are doing it well, and if so, how?
|
||||
|
||||
Iveta Lohovska:
|
||||
Yeah, I'm in the business of high stakes use cases and yes, we do those kind of projects and work, which is very exciting and interesting, and you can see the impact. So I'm in the generative AI implementation into enterprise control. An enterprise context could mean critical infrastructure, could mean telco, could mean a government, could mean intelligence organizations. So those are just a few examples, but I could flip the coin and give you an alternative for a public one where I can share, let's say a good example is climate data. And we recently worked on, on building a knowledge worker, a climate agent that is trained, of course, his foundational knowledge, because all foundational models have prior knowledge they can refer to. But the key point here is to be an expert on climate data emissions gap country cards. Every country has a commitment to meet certain reduction emission reduction goals and then benchmarked and followed through the international supervisions of the world, like the United nations environmental program and similar entities. So when you're training this agent on climate data, they're competing ideas or several sources.
|
||||
|
||||
Iveta Lohovska:
|
||||
You can source your information from the local government that is incentivized to show progress to the nation and other stakeholders faster than the actual reality, the independent entities that provide information around the state of the world when it comes to progress towards certain climate goals. And there are also different parties. So for this kind of solution, we were very lucky to work with kind of the status co provider, the benchmark around climate data, around climate publications. And what we have to ensure here is that every citation and every answer and augmentation by the generative AI on top of that is linked to the exact source of paper or publication, where it's coming from, to ensure that we can trace it back to where the climate information is coming from. If Germany performs better compared to Austria, and also the partner we work with was the United nations environmental program. So they want to make sure that they're the citadel scientific arm when it comes to giving information. And there's no compromise, could be a compromise on the structure of the answer, on the breadth and death of the information, but there should be no compromise on the exact fact fullness of the information and where it's coming from. And this is a concrete example because why, you oughta ask, why is this so important? Because it has two interfaces.
|
||||
|
||||
Iveta Lohovska:
|
||||
It has the public. You can go and benchmark your carbon footprint as an individual living in one country comparing to an individual living in another. But if you are a policymaker, which is the other interface of this application, who will write the policy recommendation of a country in their own country, or a country they're advising on, you might want to make sure that the scientific citations and the policy recommendations that you're making are correct and they are retrieved from the proper data sources. Because there will be a huge implication when you go public with those numbers or when you actually design a law that is reinforceable with legal terms and law enforcement.
|
||||
|
||||
Sabrina Aquino:
|
||||
That's very interesting, Iveta, and I think this is one of the great use cases for RAG, for example. And I think if you can talk a little bit more about how vector search is playing into all of this, how it's helping organizations do this, this.
|
||||
|
||||
Iveta Lohovska:
|
||||
Would be amazing in such specific use cases. I think the main differentiator is the traceability component, the first that you have full control on which data it will refer to, because if you deal with open source models, most of them are open, but the data it has been trained on has not been opened or given public so with vector database you introduce a step of control and explainability. Explainability means if you receive a certain answer based on your prompt, you can trace it back to the exact source where the embedding has been stored or the source of where the information is coming from and things. So this is a major use case for us for those kind of high stake solution is that you have the explainability and traceability. Explainability. It could be as simple as a semantical similarity to the text, but also the traceability of where it's coming from and the exact link of where it's coming from. So it should be, it shouldn't be referred. You can close and you can cut the line of the model referring to its previous knowledge by introducing a vector database, for example.
|
||||
|
||||
Iveta Lohovska:
|
||||
So there could be many other implications and improvements in terms of speed and just handling huge amounts of data, yet also nice to have that come with this kind of technique, but the prior use case is actually not incentivized around those.
|
||||
|
||||
Demetrios:
|
||||
So if I'm hearing you correctly, it's like yet another reason why you should be thinking about using vector databases, because you need that ability to cite your work and it's becoming a very strong design pattern. Right. We all understand now, if you can't see where this data has been pulled from or you can't get, you can't trace back to the actual source, it's hard to trust what the output is.
|
||||
|
||||
Iveta Lohovska:
|
||||
Yes, and the easiest way to kind of cluster the two groups. If you think of creative fields and marketing fields and design fields where you could go wild and crazy with the temperature on each model, how creative it could go and how much novelty it could bring to the answer are one family of use cases. But there is exactly the opposite type of use cases where this is a no go and you don't need any creativity, you just focus on, focus on the factfulness and explainability. So it's more of the speed and the accuracy of retrieving information with a high level of novelty, but not compromising on any kind of facts within the answer, because there will be legal implications and policy implications and societal implications based on the action taken on this answer, either policy recommendation or legal action. There's a lot to do with the intelligence agencies that retrieve information based on nearest neighbor or kind of a relational analysis that you can also execute with vector databases and generative AI.
|
||||
|
||||
Sabrina Aquino:
|
||||
And we know that for these high stakes sectors that data privacy is a huge concern. And when we're talking about using vector databases and storing that data somewhere, what are some of the principles or techniques that you use in terms of infrastructure, where should you store your vector database and how should you think about that part of your system?
|
||||
|
||||
Iveta Lohovska:
|
||||
Yeah, so most of the cases, I would say 99% of the cases, is that if you have such a high requirements around security and explainability, security of the data, but those security of the whole use case and environment, and the explainability and trustworthiness of the answer, then it's very natural to have expectations that will be on prem and not in the cloud, because only on prem you have a full control of where your data sits, where your model sits, the full ownership of your IP, and then the full ownership of having less question marks of the implementation and architecture, but mainly the full ownership of the end to end solution. So when it comes to those use cases, RAG on Prem, with the whole infrastructure, with the whole software and platform layers, including models on Prem, not accessible through an API, through a service somewhere where you don't know where the guardrails is, who designed the guardrails, what are the guardrails? And we see those, this a lot with, for example, copilot, a lot of question marks around that. So it's a huge part of my work is just talking of it, just sorting out that.
|
||||
|
||||
Sabrina Aquino:
|
||||
Exactly. You don't want to just give away your data to a cloud provider, because there's many implications that that comes with. And I think even your clients, they need certain certifications, then they need to make sure that nobody can access that data, something that you cannot. Exactly. I think ensure if you're just using a cloud provider somewhere, which is, I think something that's very important when you're thinking about these high stakes solutions. But also I think if you're going to maybe outsource some of the infrastructure, you also need to think about something that's similar to a hybrid cloud solution where you can keep your data and outsource the kind of management of infrastructure. So that's also a nice use case for that, right?
|
||||
|
||||
Iveta Lohovska:
|
||||
I mean, I work for HPE, so hybrid is like one of our biggest sacred words. Yeah, exactly. But actually like if you see the trends and if you see how expensive is to work to run some of those workloads in the cloud, either for training for national model or fine tuning. And no one talks about inference, inference not in ten users, but inference in hundred users with big organizations. This itself is not sustainable. Honestly, when you do the simple Linux, algebra or math of the exponential cost around this. That's why everything is hybrid. And there are use cases that make sense to be fast and speedy and easy to play with, low risk in the cloud to try.
|
||||
|
||||
Iveta Lohovska:
|
||||
But when it comes to actual GenAI work and LLM models, yeah, the answer is never straightforward when it comes to the infrastructure and the environment where you are hosting it, for many reasons, not just cost, but any other.
|
||||
|
||||
Demetrios:
|
||||
So there's something that I've been thinking about a lot lately that I would love to get your take on, especially because you deal with this day in and day out, and it is the maturity levels of the current state of Gen AI and where we are at for chat GPT or just llms and foundational models feel like they just came out. And so we're almost in the basic, basic, basic maturity levels. And when you work with customers, how do you like kind of signal that, hey, this is where we are right now, but you should be very conscientious that you're going to need to potentially work with a lot of breaking changes or you're going to have to be constantly updating. And this isn't going to be set it and forget it type of thing. This is going to be a lot of work to make sure that you're staying up to date, even just like trying to stay up to date with the news as we were talking about. So I would love to hear your take on on the different maturity levels that you've been seeing and what that looks like.
|
||||
|
||||
Iveta Lohovska:
|
||||
So I have huge exposure to GenAI for the enterprise, and there's a huge component expectation management. Why? Because chat GPT for conversational purposes and individual help is something very cool. But when this needs to be translated into actual business use cases scenario with all the constraint of the enterprise architecture, with the constraint of the use cases, the reality changes quite dramatically. So end users who are used to expect level of forgiveness as conversational chatbots have, is very different of what you will get into actual, let's say, knowledge worker type of context, or summarization type of context into the enterprise. And it's not so much to the performance of the models, but we have something called modalities of the models. And I don't think there will be ultimately one model with all the capabilities possible, let's say cult generation or image generation, voice generational, or just being very chatty and loving and so on. There will be multiple mini models out there for those. Modalities in actual architecture with reasonable cost are very difficult to handle.
|
||||
|
||||
Iveta Lohovska:
|
||||
So I would say the technical community feels we are very mature and very fast. The enterprise adoption is a totally different topic, and it's a couple of years behind, but also the society type of technologists like me, who try to keep up with the development and we know where we stand at this point, but they're the legal side and the regulations coming in, like the EU act and Biden trying to regulate the compute power, but also how societies react to this and how they adapt. And I think especially on the third one, we are far behind understanding and the implications of this technology, also adopting it at scale and understanding the vulnerabilities. That's why I enjoy so much my enterprise work is because it's a reality check. When you put the price tag attached to actual Gen AI use case in production with the inference cost and the expected performance, it's different situation when you just have an app on the phone and you chat with it and it pulls you interesting links. So yes, I think that there's a bridge to be built between the two worlds.
|
||||
|
||||
Demetrios:
|
||||
Yeah. And I find it really interesting too, because it feels to me like since it is so new, people are more willing to explore and not necessarily have that instant return of the ROI, but when it comes to more traditional ML or predictive ML, it is a bit more mature and so there's less patience for that type of exploration. Or, hey, is this use case? If you can't by now show the ROI of a predictive ML use case, then that's a little bit more dangerous. But if you can't with a Gen AI use case, it is not that big of a deal.
|
||||
|
||||
Iveta Lohovska:
|
||||
Yeah, it's basically a technology growing up in front of our eyes. It's a kind of a flying a plane while building it type of situation. We are seeing it in the real time, and I agree with you. So that the maturity around ML is one thing, but around generative AI, and they will be a model of kind of mini disappointment or decline, in my opinion, before actually maturing product. This kind of powerful technology in a sustainable way. Sustainable ways mean you can afford it, but also it proves your business case and use case. Otherwise it's just doing for the sake of doing it because everyone else is doing it.
|
||||
|
||||
Demetrios:
|
||||
Yeah, yeah, 100%. So I know we're bumping up against time here. I do feel like there was a bit of a topic that we wanted to discuss with the licenses and how that plays into basically trustworthiness and explainability. And so we were talking about how, yeah, the best is to run your own model, and it probably isn't going to be this gigantic model that can do everything. It's the, it seems like the trends are going into smaller models. And from your point of view though, we are getting new models like every week. It feels like. Yeah, especially.
|
||||
|
||||
Demetrios:
|
||||
I mean, we were just talking about this before we went live again, like databricks just released there. What is it? DBRX Yesterday you had Mistral releasing like a new base model over the weekend, and then Llama 3 is probably going to come out in the flash of an eye. So where do you stand in regards to that? It feels like there's a lot of movement in open source, but it is a little bit of, as you mentioned, like, to be cautious with the open source movement.
|
||||
|
||||
Iveta Lohovska:
|
||||
So I think it feels like there's a lot of open source, but that. So I'm totally for open sourcing and giving the people and the communities the power to be able to innovate, to do R & D in different labs so it's not locked to the view. Elite big tech companies that can afford this kind of technology. So kudos to meta for trying compared to the other equal players in the space. But open source comes with a lot of ecosystem in our world, especially for the more powerful models, which is something I don't like because it becomes like just, it immediately translates into legal fees type of conversation. It's like there are too many if else statements in those open source licensing terms where it becomes difficult to navigate, for technologists to understand what exactly this means, and then you have to bring the legal people to articulate it to you or to put additional clauses. So it's becoming a very complex environment to handle and less and less open, because there are not so many open source and small startup players that can afford to train foundational models that are powerful and useful. So it becomes a bit of a game logged to a view, and I think everyone needs to be a bit worried about that.
|
||||
|
||||
Iveta Lohovska:
|
||||
So we can use the equivalents from the past, but I don't think we are doing well enough in terms of open sourcing. The three main core components of LLM model, which is the model itself, the data it has been trained on, and the data sets, and most of the times, at least in one of those, is restricted or missing. So it's difficult space to navigate.
|
||||
|
||||
Demetrios:
|
||||
Yeah, yeah. You can't really call it trustworthy, or you can't really get the information that you need and that you would hope for if you're missing one of those three. I do like that little triangle of the necessities. So, Iveta, this has been awesome. I really appreciate you coming on here. Thank you, Sabrina, for joining us. And for everyone else that is watching, remember, don't get lost in vector space. This has been another vector space talk.
|
||||
|
||||
Demetrios:
|
||||
We are out. Have a great weekend, everyone.
|
||||
|
||||
Iveta Lohovska:
|
||||
Thank you. Bye. Thank you. Bye.
|
||||
@@ -1,15 +1,15 @@
|
||||
---
|
||||
draft: true
|
||||
draft: false
|
||||
title: How to meow on the long tail with Cheshire Cat AI? - Piero and Nicola |
|
||||
Vector Space Talks
|
||||
slug: meow-with-cheshire-cat
|
||||
short_description: Piero Savastano and Nicola Procopio discuss on the ins and
|
||||
short_description: Piero Savastano and Nicola Procopio discusses the ins and
|
||||
outs of Cheshire Cat AI.
|
||||
description: Cheshire Cat AI's Piero Savastano and Nicola Procopio discusses the
|
||||
framework's vector space complexities, community growth, and future
|
||||
cloud-based expansions.
|
||||
preview_image: /blog/from_cms/piero-and-nicola-bp-cropped.png
|
||||
date: 2024-03-18T11:00:13.338Z
|
||||
date: 2024-04-09T03:05:00.000Z
|
||||
author: Demetrios Brinkmann
|
||||
featured: false
|
||||
tags:
|
||||
@@ -19,7 +19,7 @@ tags:
|
||||
- Vector Search
|
||||
- Vector database
|
||||
---
|
||||
> *"Yes, we love Qdrant. It is our default DB. We support it in three different forms, file based, container based, and cloud based also.”*\
|
||||
> *"We love Qdrant! It is our default DB. We support it in three different forms, file based, container based, and cloud based as well.”*\
|
||||
— Piero Savastano
|
||||
>
|
||||
|
||||
@@ -31,11 +31,11 @@ Piero Savastano is the Founder and Maintainer of the open-source project, Cheshi
|
||||
|
||||
Nicola Procopio has more than 10 years of experience in data science and has worked in different sectors and markets from Telco to Healthcare. At the moment he works in the Media market, specifically on semantic search, vector spaces, and LLM applications. He has worked in the R&D area on data science projects and he has been and is currently a contributor to some open-source projects like Cheshire Cat. He is the author of popular science articles about data science on specialized blogs.
|
||||
|
||||
***Listen to the episode on Spotify, Apple Podcast, Podcast addicts, Castbox. You can also watch this episode on YouTube.***
|
||||
***Listen to the episode on [Spotify](https://open.spotify.com/episode/2d58Xui99QaUyXclIE1uuH?si=68c5f1ae6073472f), Apple Podcast, Podcast addicts, Castbox. You can also watch this episode on [YouTube](https://youtu.be/K40DIG9ZzAU?feature=shared).***
|
||||
|
||||
[embed YouTube video here]
|
||||
<iframe width="560" height="315" src="https://www.youtube.com/embed/K40DIG9ZzAU?si=rK0EVXmvNJ5OSZa4" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
|
||||
|
||||
[embed anchor.fm podcast here]
|
||||
<iframe src="https://podcasters.spotify.com/pod/show/qdrant-vector-space-talk/embed/episodes/How-to-meow-on-the-long-tail-with-Cheshire-Cat-AI----Piero-and-Nicola--Vector-Space-Talks-018-e2h7k59/a-ab31teu" height="102px" width="400px" frameborder="0" scrolling="no"></iframe>
|
||||
|
||||
## **Top takeaways:**
|
||||
|
||||
@@ -45,11 +45,11 @@ It’s time to learn how to meow! Piero in this episode of Vector Space Talks di
|
||||
|
||||
Here are the highlights from this episode:
|
||||
|
||||
1. The Art of Embedding: Discover how Cheshire Cat uses collections with an embedder, fine-tuning them through scalar quantization and other methods to enhance accuracy and performance.
|
||||
2. Vectors in Harmony: Get the lowdown on storing quantized vectors in a hybrid mode – it's all about saving memory without compromising on speed.
|
||||
3. Memory Matters: Scoop on managing different types of memory within Qdrant, the go-to vector DB for Cheshire Cat.
|
||||
4. Community Chronicles: Talking about the growing community that's shaping the evolution of Cheshire Cat - from enthusiasts to core contributors!
|
||||
5. Looking Ahead: They've got grand plans brewing for a cloud version of Cheshire Cat. Imagine a marketplace buzzing with user-generated plugins. This is the future they're painting!
|
||||
1. **The Art of Embedding:** Discover how Cheshire Cat uses collections with an embedder, fine-tuning them through scalar quantization and other methods to enhance accuracy and performance.
|
||||
2. **Vectors in Harmony:** Get the lowdown on storing quantized vectors in a hybrid mode – it's all about saving memory without compromising on speed.
|
||||
3. **Memory Matters:** Scoop on managing different types of memory within Qdrant, the go-to vector DB for Cheshire Cat.
|
||||
4. **Community Chronicles:** Talking about the growing community that's shaping the evolution of Cheshire Cat - from enthusiasts to core contributors!
|
||||
5. **Looking Ahead:** They've got grand plans brewing for a cloud version of Cheshire Cat. Imagine a marketplace buzzing with user-generated plugins. This is the future they're painting!
|
||||
|
||||
> Fun Fact: The Cheshire Cat community on Discord plays a crucial role in the development and user support of the framework, described humorously by Piero as "a mess" due to its large and active nature.
|
||||
>
|
||||
@@ -109,10 +109,10 @@ Piero Savastano:
|
||||
Dark team, you can do a lot of stuff with the framework. This is how it presents itself. We have a blog with tutorials, but going back to our numbers, it is open source, GPL licensed. We have some good numbers. We are mostly active in Italy and in a good part of Europe, East Europe, and also a little bit of our communities in the United States. There are a lot of contributors already and our docker image has been downloaded quite a few times, so it's really easy to start up and running because you just docker run and you're good to go. We have also a discord server with thousands of members. If you want to join us, it's going to be fun.
|
||||
|
||||
Piero Savastano:
|
||||
We like meme, we like to build culture around code, so it is not just the code, these are the main components of the cat. You have a chat as usual. The rabbitol is our module dedicated to document ingestion. You can extend all of these parts. We have an agent manager. Meddetter is the module to manage plugins. We have a vectordb which is Qdrant natively, by the way. We use both the file based Qdrant, the container version, and also we support the cloud version.
|
||||
We like meme, we like to build culture around code, so it is not just the code, these are the main components of the cat. You have a chat as usual. The rabbit hole is our module dedicated to document ingestion. You can extend all of these parts. We have an agent manager. Meddetter is the module to manage plugins. We have a vectordb which is Qdrant natively, by the way. We use both the file based Qdrant, the container version, and also we support the cloud version.
|
||||
|
||||
Piero Savastano:
|
||||
So if you are using Qdrant, we support the whole stack. Right now with the framework we have an embedder and a large language model coming to the embedder and language models. You can use any language model or embedded you want, closed source API, open Olama, self hosted anything. These are the main features. So the first feature of the cat is that he's ready to fight. It is already dogrized. It's model agnostic. One command in the terminal and you can meow.
|
||||
So if you are using Qdrant, we support the whole stack. Right now with the framework we have an embedder and a large language model coming to the embedder and language models. You can use any language model or embedded you want, closed source API, open Ollama, self hosted anything. These are the main features. So the first feature of the cat is that he's ready to fight. It is already dogsized. It's model agnostic. One command in the terminal and you can meow.
|
||||
|
||||
Piero Savastano:
|
||||
The other aspect is that there is not only a retrieval augmented generation system, but there is also an action agent. This is all customizable. You can plug in any script you want as an agent, or you can customize the ready default presence default agent. And one of our specialty is that we do retrieve augmented generation, not only on documents as everybody's doing, but we do also augmented generation over conversations. I can hear your keyboard. We do augmented generation over conversations and over procedures. So also our tools and form conversational forms are embedded into the DB. We have a big plugin system.
|
||||
@@ -145,10 +145,10 @@ Nicola Procopio:
|
||||
This collection with the name of the embedder used. When the user changed the embedder, we check if the embedder has the same dimension. If has the same dimension, we check also the aliases. If the aliases is the same we don't change nothing. Otherwise we create another collection and this is the drunken cut effect. The first feature that we use in the cat. Another feature is the quantization because with this Qdrant feature we improve the accuracy at the performance. We use the scalar quantitation because we are model agnostic and other quantitation like the binary quantitation.
|
||||
|
||||
Nicola Procopio:
|
||||
If you read on the Qdrant documents are experimented on not to all embedder but also for OpenAI and Coer. If I remember well with this discover quantitation and the scour quantization is used in the storage step. The vector are quantizzed and stored in a hybrid mode, the original vector on disk, the quantized vector in RAm and with this procedure we procedure we can use less memory. In case of Qdrant scalar quantization, the flat 32 elements is converted to int eight on a single number on a single element needs 75% less memory. In case of big embeddings like I don't know Gina embeddings or mistral embeddings with more than 1000 elements. This is big improvements. The second part is the retriever step. We use a quantizement query at the quantized vector to calculate causing similarity and we have the top n results like a simple semantic search pipeline.
|
||||
If you read on the Qdrant documents are experimented on not to all embedder but also for OpenAI and Coer. If I remember well with this discover quantitation and the scour quantization is used in the storage step. The vector are quantized and stored in a hybrid mode, the original vector on disk, the quantized vector in RAM and with this procedure we procedure we can use less memory. In case of Qdrant scalar quantization, the flat 32 elements is converted to int eight on a single number on a single element needs 75% less memory. In case of big embeddings like I don't know Gina embeddings or mistral embeddings with more than 1000 elements. This is big improvements. The second part is the retriever step. We use a quantizement query at the quantized vector to calculate causing similarity and we have the top n results like a simple semantic search pipeline.
|
||||
|
||||
Nicola Procopio:
|
||||
But if we want a top end results in quantiz mod, the quantity mod has less quality on the information and we use the oversampling. The oversampling is a simple multiplication. If we want top n with n ten with oversampling with a score like one five, we have 15 results, quantities results. When we have these 15 quantities results, we retrieve also the same 15 unquanted vectors. And on these unquanted vectors we rescale busset on the query and filter the best ten. This is an improvement because the retrieve step is so fast. Yes, because using these tip and tricks, the cheshire capped vectors achieve up.
|
||||
But if we want a top end results in quantize mod, the quantity mod has less quality on the information and we use the oversampling. The oversampling is a simple multiplication. If we want top n with n ten with oversampling with a score like one five, we have 15 results, quantities results. When we have these 15 quantities results, we retrieve also the same 15 unquanted vectors. And on these unquanted vectors we rescale busset on the query and filter the best ten. This is an improvement because the retrieve step is so fast. Yes, because using these tip and tricks, the Cheshire capped vectors achieve up.
|
||||
|
||||
Piero Savastano:
|
||||
Four.
|
||||
@@ -172,7 +172,7 @@ Demetrios:
|
||||
Then we do that, we can get the full program. How cool is that? Well, let's see, I'll give it another minute, let anyone from the chat ask any questions. This was really cool and I appreciate you all breaking down. Not only the space and what you're doing, but the different ways that you're using Qdrant and the challenges and the architecture behind it. I would love to know while people are typing in their questions, especially for you, Nicola, what have been some of the challenges that you've faced when you're dealing with just trying to get Cheshire Cat to be more reliable and be more able to execute with confidence?
|
||||
|
||||
Nicola Procopio:
|
||||
The challenges are in particular to mix a lot of Qdrant feature with the user needs. Because I'm a researcher, a data scientist, I like to play with strange features like binary quantization, but we need to maintain the focus on the user needs, on the user behavior. And sometimes we cut some feature on the Chichircat because it's not important now for for the user and we can introduce some bug, or rather misunderstanding for the user.
|
||||
The challenges are in particular to mix a lot of Qdrant feature with the user needs. Because I'm a researcher, a data scientist, I like to play with strange features like binary quantization, but we need to maintain the focus on the user needs, on the user behavior. And sometimes we cut some feature on the Cheshire cat because it's not important now for for the user and we can introduce some bug, or rather misunderstanding for the user.
|
||||
|
||||
Demetrios:
|
||||
Can you hear me? Yeah. All right, good. Now I'm seeing a question come through in the chat that is asking if you are thinking about cloud version of the cat. Like a SaaS, it's going to come. It's in the works.
|
||||
@@ -190,7 +190,7 @@ Demetrios:
|
||||
Yeah, that's the best. That is really cool. Simone is asking if there's companies that are already using Cheshire cat, and if you can mention a few.
|
||||
|
||||
Piero Savastano:
|
||||
Yeah, okay. In Italy, there are at least 1015 companies distributed along education, customer care, typical chatbot usage. Also, one of them in particular is trying to build for public administration, which is really hard to do on the international level. We are seeing something in Germany, like web agencies starting to use the cat a little on the USA. Mostly they are trying to build agents using the cat and Olama as a runner. And a company in particular presented in a conference in Vegas a pitch about a 3d avatar. Inside the avatar, there is the cat as a linguistic device.
|
||||
Yeah, okay. In Italy, there are at least 1015 companies distributed along education, customer care, typical chatbot usage. Also, one of them in particular is trying to build for public administration, which is really hard to do on the international level. We are seeing something in Germany, like web agencies starting to use the cat a little on the USA. Mostly they are trying to build agents using the cat and Ollama as a runner. And a company in particular presented in a conference in Vegas a pitch about a 3d avatar. Inside the avatar, there is the cat as a linguistic device.
|
||||
|
||||
Demetrios:
|
||||
Oh, nice.
|
||||
@@ -199,7 +199,7 @@ Piero Savastano:
|
||||
To be honest, we have a little problem tracking companies because we still have no telemetry. We decided to be no telemetry for the moment. So I hope companies will contribute and make themselves happen. If that does not, we're going to track a little more. But companies using the cat are at least in the 50, 60, 70.
|
||||
|
||||
Demetrios:
|
||||
Yeah, nice. So if anybody out there is using the cat, and you have not talked to Piro yet, let him know so that he can have a good idea of what you're doing and how you're doing it. There's also another question coming through about the market analysis. Are there some competitors?
|
||||
Yeah, nice. So if anybody out there is using the cat, and you have not talked to Piero yet, let him know so that he can have a good idea of what you're doing and how you're doing it. There's also another question coming through about the market analysis. Are there some competitors?
|
||||
|
||||
Piero Savastano:
|
||||
There are many competitors. When you go down to what distinguishes the cat from many other frameworks that are coming out, we decided since the beginning to go for a plugin based operational agent. And at the moment, most frameworks are retrieval augmented generation frameworks. We have both retrieval augmented generation. We have tooling, we have forms. The tools and the forms are also embedded. So the cat can have 20,000 tools, because we also embed the tools and we make a recall over the function calling. So we scaled up both documents, conversation and tools, conversational forms, and I've not seen anybody doing that till now.
|
||||
|
||||
@@ -0,0 +1,263 @@
|
||||
---
|
||||
draft: true
|
||||
title: "Teaching Vector Databases at Scale - Alfredo Deza | Vector Space Talks #019"
|
||||
slug: teaching-vector-db-at-scale
|
||||
short_description: Alfredo Deza tackles AI teaching, the intersection of
|
||||
technology and academia, and the value of consistent learning.
|
||||
description: Alfredo Deza discusses the practicality of machine learning
|
||||
operations, highlighting how personal interest in topics like wine datasets
|
||||
enhances engagement, while reflecting on the synergies between his athletic
|
||||
discipline and the persistent, straightforward approach required for
|
||||
effectively educating on vector databases and large language models.
|
||||
preview_image: /blog/from_cms/alfredo-deza-bp-cropped.png
|
||||
date: 2024-04-02T22:57
|
||||
author: Demetrios Brinkmann
|
||||
featured: false
|
||||
tags:
|
||||
- Vector Search
|
||||
- Retrieval Augmented Generation
|
||||
- Vector Space Talks
|
||||
- Coursera
|
||||
---
|
||||
> *"So usually I get asked, why are you using Qdrant? What's the big deal? Why are you picking these over all of the other ones? And to me it boils down to, aside from being renowned or recognized, that it works fairly well. There's one core component that is critical here, and that is it has to be very straightforward, very easy to set up so that I can teach it, because if it's easy, well, sort of like easy to or straightforward to teach, then you can take the next step and you can make it a little more complex, put other things around it, and that creates a great development experience and a learning experience as well.”*\
|
||||
— Alfredo Deza
|
||||
>
|
||||
|
||||
Alfredo is a software engineer, speaker, author, and former Olympic athlete working in Developer Relations at Microsoft. He has written several books about programming languages and artificial intelligence and has created online courses about the cloud and machine learning.
|
||||
|
||||
He currently is an Adjunct Professor at Duke University, and as part of his role, works closely with universities around the world like Georgia Tech, Duke University, Carnegie Mellon, and Oxford University where he often gives guest lectures about technology.
|
||||
|
||||
***Listen to the episode on [Spotify](https://open.spotify.com/episode/4HFSrTJWxl7IgQj8j6kwXN?si=99H-p0fKQ0WuVEBJI9ugUw), Apple Podcast, Podcast addicts, Castbox. You can also watch this episode on [YouTube](https://youtu.be/3l6F6A_It0Q?feature=shared).***
|
||||
|
||||
<iframe width="560" height="315" src="https://www.youtube.com/embed/3l6F6A_It0Q?si=cFZGAh7995iHilcY" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
|
||||
|
||||
<iframe src="https://podcasters.spotify.com/pod/show/qdrant-vector-space-talk/embed/episodes/Teaching-Vector-Databases-at-Scale---Alfredo-Deza--Vector-Space-Talks-019-e2hhjlo/a-ab3qp7u" height="102px" width="400px" frameborder="0" scrolling="no"></iframe>
|
||||
|
||||
## **Top takeaways:**
|
||||
|
||||
How does a former athlete such as Alfredo Deza end up in this AI and Machine Learning industry? That’s what we’ll find out in this episode of Vector Space Talks. Let’s understand how his background as an olympian offers a unique perspective on consistency and discipline that's a real game-changer in this industry.
|
||||
|
||||
Here are some things you’ll discover from this episode:
|
||||
|
||||
1. **The Intersection of Teaching and Tech:** Alfredo discusses on how to effectively bridge the gap between technical concepts and student understanding, especially when dealing with complex topics like vector databases.
|
||||
2. **Simplified Learning:** Dive into Alfredo's advocacy for simplicity in teaching methods, mirroring his approach with Qdrant and the potential for a Rust in-memory implementation aimed at enhancing learning experiences.
|
||||
3. **Beyond the Titanic Dataset:** Discover why Alfredo prefers to teach with a wine dataset he developed himself, underscoring the importance of using engaging subject matter in education.
|
||||
4. **AI Learning Acceleration:** Alfredo discusses the struggle universities face to keep pace with AI advancements and how online platforms can offer a more up-to-date curriculum.
|
||||
5. **Consistency is Key:** Alfredo draws parallels between the discipline required in high-level athletics and the ongoing learning journey in AI, zeroing in on his mantra, “There is no secret” to staying consistent.
|
||||
|
||||
> Fun Fact: Alfredo tells the story of athlete Dick Fosbury's invention of the Fosbury Flop to highlight the significance of teaching simplicity.
|
||||
>
|
||||
|
||||
## Show notes:
|
||||
|
||||
00:00 Teaching machine learning, Python to graduate students.\
|
||||
06:03 Azure AI search service simplifies teaching, Qdrant facilitates learning.\
|
||||
10:49 Controversy over high jump style.\
|
||||
13:18 Embracing past for inspiration, emphasizing consistency.\
|
||||
15:43 Consistent learning and practice lead to success.\
|
||||
20:26 Teaching SQL uses SQLite, Rust has limitations.\
|
||||
25:21 Online platforms improve and speed up education.\
|
||||
29:24 Duke and Coursera offer specialized language courses.\
|
||||
31:21 Passion for wines, creating diverse dataset.\
|
||||
35:00 Encouragement for vector db discussion, wrap up.
|
||||
|
||||
## More Quotes from Alfredo:
|
||||
|
||||
*"Qdrant makes it straightforward. We use it in-memory for my classes and I would love to see something similar setup in Rust to make teaching even easier.”*\
|
||||
— Alfredo Deza
|
||||
|
||||
*"Retrieval augmented generation is kind of like having an open book test. So the large language model is the student, and they have an open book so they can see the answers and then repackage that into their own words and provide an answer.”*\
|
||||
— Alfredo Deza
|
||||
|
||||
*"With Qdrant, I appreciate that the use of the Python API is so simple. It avoids the complexity that comes from having a back-end system like in Rust where you need an actual instance of the database running.”*\
|
||||
— Alfredo Deza
|
||||
|
||||
## Transcript:
|
||||
Demetrios:
|
||||
What is happening? Everyone, welcome back to another vector space talks. I am Demetrios, and I am joined today by good old Sabrina. Where you at, Sabrina? Hello?
|
||||
|
||||
Sabrina Aquino:
|
||||
Hello, Demetrios. I'm from Brazil. I'm in Brazil right now. I know that you are traveling currently.
|
||||
|
||||
Demetrios:
|
||||
Where are you? At Kubecon in Paris. And it has been magnificent. But I could not wait to join the session today because we've got Alfredo coming at us.
|
||||
|
||||
Alfredo Deza:
|
||||
What's up, dude? Hi. How are you?
|
||||
|
||||
Demetrios:
|
||||
I'm good, man. It's been a while. I think the last time that we chatted was two years ago, maybe right before your book came out. When did the book come out?
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, something like that. I would say a couple of years ago. Yeah. I wrote, co authored practical machine learning operations with no gift. And it was published on O'Reilly.
|
||||
|
||||
Demetrios:
|
||||
Yeah. And that was, I think, two years ago. So you've been doing a lot of stuff since then. Let's be honest, you are maybe one of the most active men on the Internet. I always love seeing what you're doing. You're bringing immense value to everything that you touch. I'm really excited to be able to chat with you for this next 30 minutes.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, of course.
|
||||
|
||||
Demetrios:
|
||||
Maybe just, we'll start it off. We're going to get into it when it comes to what you're doing and really what the space looks like right now. Right. But I would love to hear a little bit of what you've been up to since, for the last two years, because I haven't talked to you.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, that's right. Well, several different things, actually. Right after we chatted last time, I joined Microsoft to work in developer relations. Microsoft has a big group of folks working in developer relations. And basically, for me, it signaled my shift away from regular software engineering. I was primarily doing software engineering and thought that perhaps with the books and some of the courses that I had published, it was time for me to get into more teaching and providing useful content, which is really something very rewarding. And in developer relations, in advocacy in general, it's kind of like a way of teaching. We demonstrate technology, how it works from a technical point of view.
|
||||
|
||||
Alfredo Deza:
|
||||
So aside from that, started working really closely with several different universities. I work with Georgia Tech, Oxford University, Carnegie Mellon University, and Duke University, where I've been working as an adjunct professor for a couple of years as well. So at Duke, what I do is I teach a couple of classes a year. One is on machine learning. Last year was machine learning operations, and this year it's going to, I think, hopefully I'm not messing anything up. I think we're going to shift a little bit to doing operations with large language models. And in the fall I teach a programming class for graduate students that want to join one of the graduate programs and they want to get a primer on Python. So I teach a little bit of that.
|
||||
|
||||
Alfredo Deza:
|
||||
And in the meantime, also in partnership with Duke, getting a lot of courses out on Coursera, and from large language models to doing stuff with Azure, to machine learning operations, to rust, I've been doing a lot of rust lately, which I really like. So, yeah, so a lot of different things, but I think the core pillar for me remains being able to teach and spread the knowledge.
|
||||
|
||||
Demetrios:
|
||||
Love it, man. And I know you've been diving into vector databases. Can you tell us more?
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, well, the thing is that when you're trying to teach, and yes, one of the courses that we had out for large language models was applying retrieval augmented generation, which is the basis for vector databases, to see how it works. This is how it works. These are the components that you need. Let's create an application from scratch and see how it works. And for those that don't know, retrieval augmented generation is kind of like having. The other day I saw a description about this, which I really like, which is a way of, it's kind of like having an open book test. So the large language model is the student, and they have an open book so they can see the answers and then repackage that into their own words and provide an answer, which is kind of like what we do with vector databases in the retrieval augmented generation pattern. We've been putting a lot of examples on how to do these, and in the case of Azure, you're enabling certain services.
|
||||
|
||||
Alfredo Deza:
|
||||
There's the Azure AI search service, which is really good. But sometimes when you're trying to teach specifically, it is useful to have a very straightforward way to do this and applying or creating a retrieval augmented generation pattern, it's kind of tricky, I think. We're not there yet to do it in a nice, straightforward way. So there are several different options, Qdrant being one of them. So usually I get asked, why are you using Qdrant? What's the big deal? Why are you picking these over all of the other ones? And to me it boils down to, aside from being renowned or recognized, that it works fairly well. There's one core component that is critical here, and that is it has to be very straightforward, very easy to set up so that I can teach it, because if it's easy, well, sort of like easy to or straightforward to teach, then you can take the next step and you can make it a little more complex, put other things around it, and that creates a great development experience and a learning experience as well. If something is very complex, if the list of requirements is very long, you're not going to be very happy, you're going to spend all this time trying to figure, and when you have, similar to what happens with automation, when you have a list of 20 different things that you need to, in order to, say, deploy a website, you're going to get things out of order, you're going to forget one thing, you're going to have a typo, you're going to mess it up, you're going to have to start from scratch, and you're going to get into a situation where you can't get out of it. And Qdrant does provide a very straightforward way to run the database, and that one is the in memory implementation with Python.
|
||||
|
||||
Alfredo Deza:
|
||||
So you can actually write a little bit of python once you install the libraries and say, I want to instantiate a vector database and I wanted to run it in memory. So for teaching, this is great. It's like, hey, of course it's not for production, but just write these couple of lines and let's get right into it. Let's just start populating these and see how it works. And it works. It's great. You don't need to have all of these, like, wow, let's launch Kubernetes over here and let's have all of these dynamic. No, why? I mean, sure, you want to create a business model and you want to launch to production eventually, and you want to have all that running perfect.
|
||||
|
||||
Alfredo Deza:
|
||||
But for this setup, like for understanding how it works, for trying baby steps into understanding vector databases, this is perfect. My one requirement, or my one wish list item is to have that in memory thing for rust. That would be pretty sweet, because I think it'll make teaching rust and retrieval augmented generation with rust much easier. I wouldn't have to worry about bringing up containers or external services. So that's the deal with rust. And I'll tell you one last story about why I think specifically making it easy to get started with so that I can teach it, so that others can learn from it, is crucial. I would say almost 50 years ago, maybe a little bit more, my dad went to Italy to have a course on athletics. My dad was involved in sports and he was going through this, I think it was like a six month specialization on athletics.
|
||||
|
||||
Alfredo Deza:
|
||||
And he was in class and it had been recent that the high jump had transitioned from one style to the other. The previous style, the old style right now is the old style. It's kind of like, it was kind of like over the bar. It was kind of like a weird style. And it had recently transitioned to a thing called the Fosbury flop. This person, his last name is Dick Fosbury, invented the Fosbury flop. He said, no, I'm just going to go straight at it, then do a little curve and then jump over it. And then he did, and then he started winning everything.
|
||||
|
||||
Alfredo Deza:
|
||||
And everybody's like, what this guy? Well, first they thought he was crazy, and they thought that dismissive of what he was trying to do. And there were people that sticklers that wanted to stay with the older style, but then he started beating records and winning medals, and so people were like, well, is this a good thing? Let's try it out. So there was a whole. They were casting doubt. It's like, is this really the thing? Is this really what we should be doing? So one of the questions that my dad had to answer in this specialization he did in Italy was like, which style is better, it's the old style or the new style? And so my dad said, it's the new style. And they asked him, why is the new style better? And he didn't choose the path of answering the, well, because this guy just won the Olympics or he just did a record over here that at the end is meaningless. What he said was, it is the better style because it's easier to teach and it is 100% correct. When you're teaching high jump, it is much easier to teach the Fosbury flop than the other style.
|
||||
|
||||
Alfredo Deza:
|
||||
It is super hard. So you start seeing this parallel in teaching and learning where, but with this one, you have all of these world records and things are going great. Well, great. But is anybody going to try, are you going to have more people looking into it or are you going to have less? What is it that we're trying to do here? Right.
|
||||
|
||||
Demetrios:
|
||||
Not going to lie, I did not see how you were going to land the plane on coming from the high jump into the vector database space, but you did it gracefully. That was well done. So, basically, the easier it is to teach, the more people are going to be able to jump on board and the more people are going to be able to get value out of it.
|
||||
|
||||
Sabrina Aquino:
|
||||
I absolutely love it, by the way. It's a pleasure to meet you, Alfredo. And I was actually about to ask you. I love your background as an olympic athlete. Right. And I was wondering, do you make any connections or how do we interact this background with your current teaching and AI? And do you see any similarities or something coming from that approach into what you've applied?
|
||||
|
||||
Alfredo Deza:
|
||||
Well, you're bringing a great point. It's taken me a very long time to feel comfortable talking about my professional sports past. I don't want to feel like I'm overwhelming anyone or trying to be like a show off. So I usually try not to mention, although I'm feeling more comfortable mentioning my professional past. But the only situations where I think it's good to talk about it is when I feel like there's a small chance that I might get someone thinking about the possibilities of what they can actually do and what they can try. And things that are seemingly complex might be achievable. So you mentioned similarities, but I think there are a couple of things that happen when you're an athlete in any sport, really, that you're trying to or you're operating at the very highest level and there's several things that happen there. You have to be consistent.
|
||||
|
||||
Alfredo Deza:
|
||||
And it's something that I teach my kids as well. I have one of my kids, he's like, I did really a lot of exercise today and then for a week he doesn't do anything else. And he's like, now I'm going to do exercise again. And she's going to do 4 hours. And it's like, wait a second, wait a second. It's okay. You want to do it. This is great.
|
||||
|
||||
Alfredo Deza:
|
||||
But no intensity. You need to be consistent. Oh, dad, you don't let me work out and it's like, no work out. Good, I support you, but you have to be consistent and slowly start ramping up and slowly start getting better. And it happens a lot with learning. We are in an era that concepts and things are advancing so fast that things are getting obsolete even faster. So you're always in this motion of trying to learn. So what I would say is the similarities are in the consistency.
|
||||
|
||||
Alfredo Deza:
|
||||
You have to keep learning, you have to keep applying yourself. But it can be like, oh, today I'm going to read this whole book from start to end and you're just going to learn everything about, I don't know, rust. It's like, well, no, try applying rust a little bit every day and feel comfortable with it. And at the very end you will do better. Like, you can't go with high intensity because you're going to get burned out, you're going to overwhelmed and it's not going to work out. You don't go to the Olympics by working out for like a few months. Actually, a very long time ago, a reporter asked me, how many months have you been working out preparing for the Olympics? It's like, what do you mean with how many months? I've been training my whole life for this. What are we talking about?
|
||||
|
||||
Demetrios:
|
||||
We're not talking in months or years. We're talking in lifetimes, right?
|
||||
|
||||
Alfredo Deza:
|
||||
So you have to take it easy. You can't do that. And beyond that, consistency. Consistency goes hand in hand with discipline. I came to the US in 2006. I don't live like I was born in Peru and I came to the US with no degree. I didn't go to college. Well, I went to college for a few months and then I dropped out and I didn't have a career, I didn't have experience.
|
||||
|
||||
Alfredo Deza:
|
||||
I was just recently married. I have never worked in my life because I used to be a professional athlete. And the only thing that I decided to do was to do amazing work, apply myself and try to keep learning and never stop learning. In the back of my mind, it's like, oh, I have a tremendous knowledge gap that I need to fulfill by learning. And actually, I have tremendous respect and I'm incredibly grateful by all of the people that opened doors for me and gave me an opportunity, one of them being Noah Giff, which I co authored a few books with him and some of the courses. And he actually taught me to write Python. I didn't know how to program. And he said, you know what? I think you should learn to write some python.
|
||||
|
||||
Alfredo Deza:
|
||||
And I was like, python? Why would I ever need to do that? And I did. He's like, let's just find something to automate. I mean, what a concept. Find something to apply automation. And every week on Fridays, we'll just take a look at it and that's it. And we did that for a while. And then he said, you know what? You should apply for speaking at Python. How can I be speaking at a conference when I just started learning? It's like your perspective is different.
|
||||
|
||||
Alfredo Deza:
|
||||
You just started learning these. You're going to do it in an interesting way. So I think those are concepts that are very important to me. Stay disciplined, stay consistent, and keep at it. The secret is that there's no secret. That's the bottom line. You have to keep consistent. Otherwise things are always making excuses.
|
||||
|
||||
Alfredo Deza:
|
||||
Is very simple.
|
||||
|
||||
Demetrios:
|
||||
The secret is there is no secret. That is beautiful. So you did kind of sprinkle this idea of, oh, I wish there was more stuff happening with Qdrant and rust. Can you talk a little bit more to that? Because one piece of Qdrant that people tend to love is that it's built in rust. Right. But also, I know that you mentioned before, could we get a little bit of this action so that I don't have to deal with any. What was it you were saying? The containers.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah. Right. Now, if you want to have a proof of concept, and I always go for like, what's the easiest, the most straightforward, the less annoying things I need to do, the better. And with Python, the Python API for Qdrant, you can just write a few lines and say, I want to create an instance in memory and then that's it. The database is created for you. This is very similar, or I would say actually almost identical to how you run SQLite. Sqlite is the embedded database you can create in memory. And it's actually how I teach SQL as well.
|
||||
|
||||
Alfredo Deza:
|
||||
When I have to teach SQl, I use sqlite. I think it's perfect. But in rust, like you said, Qdrant's backend is built on rust. There is no in memory implementation. So you are required to have an actual instance of the Qdrant database running. So you have a couple of options, but one of them probably means you'll have to bring up a container with Qdrant running and then you'll have to connect to that instance. So when you're teaching, the development environments are kind of constrained. Either you are in a lab somewhere like Crusader has labs, but those are self contained.
|
||||
|
||||
Alfredo Deza:
|
||||
It's kind of tricky to get them running 100%. You can run multiple containers at the same time. So things start becoming more complex. Not only more complex for the learner, but also in this case, like the teacher, me who wants to figure out how to make this all run in a very constrained environment. And that makes it tricky. And I fasted the team, by the way, and I was told that maybe at some point they can do some magic and put the in memory implementation on the rust side of things, which I think it would be tremendous.
|
||||
|
||||
Sabrina Aquino:
|
||||
We're going to advocate for that on our side. We're also going to be asking for it. And I think this is really good too. It really makes it easier. Me as a student not long ago, I do see what you mean. It's quite hard to get it all working very fast in the time of a class that you don't have a lot of time and students can get. I don't know, it's quite complex. I do get what you mean.
|
||||
|
||||
Sabrina Aquino:
|
||||
And you also are working both on the tech industry and on academia, which I think is super interesting. And I always kind of feel like those two are a bit disconnected sometimes. And I was wondering what you think that how important is the collaboration of these two areas considering how fast the AI space is going to right now? And what are your thoughts?
|
||||
|
||||
Alfredo Deza:
|
||||
Well, I don't like generalizing, but I'm going to generalize right now. I would say most universities are several steps behind, and there's a lot of complexities involved in higher education specifically. Most importantly, these institutions tend to be fairly large, and with fairly large institutions, what do you get? Oh, you get the magical bureaucracy for anything you want to do. Something like, oh, well, you need to talk to that department that needs to authorize something, that needs to go to some other department, and it's like, I'm going to change the curriculum. It's like, no, you can't. What does that mean? I have actually had conversations with faculty in universities where they say, listen, curricula. Yeah, we get that. We need to update it, but we change curricula every five years.
|
||||
|
||||
Alfredo Deza:
|
||||
And so. See you in a while. It's been three years. We have two more years to go. See you in a couple of years. And that's detrimental to students now. I get it. Building curricula, it's very hard.
|
||||
|
||||
Alfredo Deza:
|
||||
It takes a lot of work for the faculty to put something together. So it is something that, from a faculty perspective, it's like they're not going to get paid more if they update the curriculum.
|
||||
|
||||
Demetrios:
|
||||
Right.
|
||||
|
||||
Alfredo Deza:
|
||||
And it's a massive amount of work now that, of course, comes to the detriment of the learner. The student will be under service because they will have to go through curricula that is fairly dated. Now, there are situations and there are programs where this doesn't happen. And Duke, I've worked with several. They're teaching Llama file, which was built by Mozilla. And when did Llama file came out? It was just like a few months ago. And I think it's incredible. And I think those skills that are the ones that students need today in order to not only learn these things, but also be able to apply them when they're looking for a job or trying to professionally even apply them into their day to day, now that's one side of things.
|
||||
|
||||
Alfredo Deza:
|
||||
But there's the other aspect. In the case of Duke, as well as other universities out there, they're using these online platforms so that they can put courses out there faster. Do you really need to go through a four year program to understand how retrieval augmented generation works? Or how to implement it? I would argue no, but would you be better out, like, taking a course that will take you perhaps a couple of weeks to go through and be fairly proficient? I would say yes, 100%. And you see several institutions putting courses out there that are meaningful, that are useful, that they can cope with the speed at which things are needed. I think it's kind of good. And I think that sometimes we tend to think about knowledge and learning things, kind of like in a bubble, especially here in the US. I think there's this college is this magical place where all of the amazing things happen. And if you don't go to college, things are going to go very bad for you.
|
||||
|
||||
Alfredo Deza:
|
||||
And I don't think that's true. I think if you like college, if you like university, by all means take advantage of it. You want to experience it. That sounds great. I think there's tons of opportunity to do it outside of the university or the college setting and taking online courses from validated instructors. They have a good profile. Not someone that just dumped something on genetic AI and started.
|
||||
|
||||
Demetrios:
|
||||
Someone like you.
|
||||
|
||||
Alfredo Deza:
|
||||
Well, if you want to. Yeah, sure, why not? I mean, there's students that really like my teaching style. I think that's great. If you don't like my teaching style. Sometimes I tend to go a little bit slower because I don't want to overwhelm anyone. That's all good. But there is opportunity. And when I mention these things, people are like, oh, really? I'm not advertising for Coursera or anything else, but some of these platforms, if you pay a monthly fee, I think it's between $40 and $60.
|
||||
|
||||
Alfredo Deza:
|
||||
I think on the expensive side, you can take advantage of all of these courses and as much as you can take them. Sometimes even companies say, hey, you have a paid subscription, go take it all. And I've met people like that. It's like, this is incredible. I'm learning so much. Perfect. I think there's a mix of things. I don't think there's like a binary answer, like, oh, you need to do this, or, no, don't do that, and everything's going to be well again.
|
||||
|
||||
Demetrios:
|
||||
Yeah. Can you talk a little bit more about your course? And if I wanted to go on Coursera, what can I expect from.
|
||||
|
||||
Alfredo Deza:
|
||||
You know, and again, I don't think as much as I like talking about my courses and the things that I do, I want to emphasize, like, if someone is watching this video or listening into what we're talking about, find something that is interesting to you and find a course that kind of delivers that thing, that sliver of interesting stuff, and then try it out. I think that's the best way. Don't get overwhelmed by. It's like, is this the right vector database that I should be learning? Is this instructor? It's like, no, try it out. What's going to happen? You don't like it when you're watching a bad video series or docuseries on Netflix or any streaming platform? Do you just like, I pay my $10 a month, so I'm going to muster through this whole 20 more episodes of this thing that I don't like. It's meaningless. It doesn't matter. Just move on.
|
||||
|
||||
Alfredo Deza:
|
||||
So having said that, on Coursera specifically with Duke University, we tend to put courses out there that are going to be used in our programs in the things that I teach. For example, we just released the large language models. Specialization and specialization is a grouping of between four and six courses. So in there we have doing large language models with Azure, for example, introduction to generative AI, having a very simple rag pattern with Qdrant. I also have examples on how to do it with Azure AI search, which I think is pretty cool as well. How to do it locally with Llama file, which I think is great. You can have all of these large language models running locally, and then you have a little bit of Qdrant sprinkle over there, and then you have rack pattern. Now, I tend to teach with things that I really like, and I'll give you a quick example.
|
||||
|
||||
Alfredo Deza:
|
||||
I think there's three data sets that are one of the top three most used data sets in all of machine learning and data science. Those are the Boston housing market, the diabetes data set in the US, and the other one is the Titanic. And everybody uses those. And I don't really understand why. I mean, perhaps I do understand why. It's because they're easy, they're clean, they're ready to go. Nothing's ever wrong with these, and everybody has used them to boredom. But for the life of me, you wouldn't be able to convince me to use any of those, because these are not topics that I really care about and they don't resonate with me.
|
||||
|
||||
Alfredo Deza:
|
||||
The Titanic specifically is just horrid. Well, if I was 37 and I'm on first class and I'm male, would I survive? It's like, what are we trying to do here? How is this useful to anyone? So I tend to use things that I like, and I'm really passionate about wine. So I built my own data set, which is a collection of wines from all over the world, they have the ratings, they have the region, they have the type of grape and the notes and the name of the wine. So when I'm teaching them, like, look at this, this is amazing. It's wines from all over the world. So let's do a little bit of things here. So, for rag, what I was able to do is actually in the courses as well. I do, ah, I really know wines from Argentina, but these wines, it would be amazing if you can find me not a Malbec, but perhaps a cabernet franc.
|
||||
|
||||
Alfredo Deza:
|
||||
That is amazing. From, it goes through Qdrant, goes back to llama file using some large language model or even small language model, like the Phi 2 from Microsoft, I think is really good. And he goes, it tells. Yeah, sure. I get that you want to have some good wines. Here's some good stuff that I can give you. And so it's great, right? I think it's great. So I think those kinds of things that are interesting to the person that is teaching or presenting, I think that's the key, because whenever you're talking about things that are very boring, that you do not care about, things are not going to go well for you.
|
||||
|
||||
Alfredo Deza:
|
||||
I mean, if I didn't like teaching, if I didn't like vector databases, you would tell right away. It's like, well, yes, I've been doing stuff with the vector databases. They're good. Yeah, Qdrant, very good. You would tell right away. I can't lie. Very good.
|
||||
|
||||
Demetrios:
|
||||
You can't fool anybody.
|
||||
|
||||
Alfredo Deza:
|
||||
No.
|
||||
|
||||
Demetrios:
|
||||
Well, dude, this is awesome. We will drop a link to the chat. We will drop a link to the course in the chat so that in case anybody does want to go on this wine tasting journey with you, they can. And I'm sure there's all kinds of things that will spark the creativity of the students as they go through it, because when you were talking about that, I was like, oh, it would be really cool to make that same type of thing, but with ski resorts there, you go around the world. And if I want this type of ski resort, I'm going to just ask my chat bot. So I'm excited to see what people create with it. I also really appreciate you coming on here, giving us your time and talking through all this. It's been a pleasure, as always, Alfredo.
|
||||
|
||||
Demetrios:
|
||||
Thank you so much.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, thank you. Thank you for having me. Always happy to chat with you. I think Qdrant is doing a very solid product. Hopefully, my wish list item of in memory in rust comes to fruition, but I get it. Sometimes there are other priorities. It's all good. Yeah.
|
||||
|
||||
Alfredo Deza:
|
||||
If anyone wants to connect with me, I'm always active on LinkedIn primarily. Always happy to connect with folks and talk about learning and improving and always being a better person.
|
||||
|
||||
Demetrios:
|
||||
Excellent. Well, we will sign off, and if anyone else out there wants to come on here and talk to us about vector databases, we're always happy to have you. Feel free to reach out. And remember, don't get lost in vector space, folks. We will see you on the next one.
|
||||
|
||||
Sabrina Aquino:
|
||||
Good night. Thank you so much.
|
||||
@@ -0,0 +1,264 @@
|
||||
---
|
||||
draft: false
|
||||
title: Teaching Vector Databases at Scale - Alfredo Deza | Vector Space Talks
|
||||
slug: teaching-vector-db-at-scale
|
||||
short_description: Alfredo Deza tackles AI teaching, the intersection of
|
||||
technology and academia, and the value of consistent learning.
|
||||
description: Alfredo Deza discusses the practicality of machine learning
|
||||
operations, highlighting how personal interest in topics like wine datasets
|
||||
enhances engagement, while reflecting on the synergies between his
|
||||
professional sportsman discipline and the persistent, straightforward approach
|
||||
required for effectively educating on vector databases and large language
|
||||
models.
|
||||
preview_image: /blog/from_cms/alfredo-deza-bp-cropped.png
|
||||
date: 2024-04-09T03:06:00.000Z
|
||||
author: Demetrios Brinkmann
|
||||
featured: false
|
||||
tags:
|
||||
- Vector Search
|
||||
- Retrieval Augmented Generation
|
||||
- Vector Space Talks
|
||||
- Coursera
|
||||
---
|
||||
> *"So usually I get asked, why are you using Qdrant? What's the big deal? Why are you picking these over all of the other ones? And to me it boils down to, aside from being renowned or recognized, that it works fairly well. There's one core component that is critical here, and that is it has to be very straightforward, very easy to set up so that I can teach it, because if it's easy, well, sort of like easy to or straightforward to teach, then you can take the next step and you can make it a little more complex, put other things around it, and that creates a great development experience and a learning experience as well.”*\
|
||||
— Alfredo Deza
|
||||
>
|
||||
|
||||
Alfredo is a software engineer, speaker, author, and former Olympic athlete working in Developer Relations at Microsoft. He has written several books about programming languages and artificial intelligence and has created online courses about the cloud and machine learning.
|
||||
|
||||
He currently is an Adjunct Professor at Duke University, and as part of his role, works closely with universities around the world like Georgia Tech, Duke University, Carnegie Mellon, and Oxford University where he often gives guest lectures about technology.
|
||||
|
||||
***Listen to the episode on [Spotify](https://open.spotify.com/episode/4HFSrTJWxl7IgQj8j6kwXN?si=99H-p0fKQ0WuVEBJI9ugUw), Apple Podcast, Podcast addicts, Castbox. You can also watch this episode on [YouTube](https://youtu.be/3l6F6A_It0Q?feature=shared).***
|
||||
|
||||
<iframe width="560" height="315" src="https://www.youtube.com/embed/3l6F6A_It0Q?si=cFZGAh7995iHilcY" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
|
||||
|
||||
<iframe src="https://podcasters.spotify.com/pod/show/qdrant-vector-space-talk/embed/episodes/Teaching-Vector-Databases-at-Scale---Alfredo-Deza--Vector-Space-Talks-019-e2hhjlo/a-ab3qp7u" height="102px" width="400px" frameborder="0" scrolling="no"></iframe>
|
||||
|
||||
## **Top takeaways:**
|
||||
|
||||
How does a former athlete such as Alfredo Deza end up in this AI and Machine Learning industry? That’s what we’ll find out in this episode of Vector Space Talks. Let’s understand how his background as an olympian offers a unique perspective on consistency and discipline that's a real game-changer in this industry.
|
||||
|
||||
Here are some things you’ll discover from this episode:
|
||||
|
||||
1. **The Intersection of Teaching and Tech:** Alfredo discusses on how to effectively bridge the gap between technical concepts and student understanding, especially when dealing with complex topics like vector databases.
|
||||
2. **Simplified Learning:** Dive into Alfredo's advocacy for simplicity in teaching methods, mirroring his approach with Qdrant and the potential for a Rust in-memory implementation aimed at enhancing learning experiences.
|
||||
3. **Beyond the Titanic Dataset:** Discover why Alfredo prefers to teach with a wine dataset he developed himself, underscoring the importance of using engaging subject matter in education.
|
||||
4. **AI Learning Acceleration:** Alfredo discusses the struggle universities face to keep pace with AI advancements and how online platforms can offer a more up-to-date curriculum.
|
||||
5. **Consistency is Key:** Alfredo draws parallels between the discipline required in high-level athletics and the ongoing learning journey in AI, zeroing in on his mantra, “There is no secret” to staying consistent.
|
||||
|
||||
> Fun Fact: Alfredo tells the story of athlete Dick Fosbury's invention of the Fosbury Flop to highlight the significance of teaching simplicity.
|
||||
>
|
||||
|
||||
## Show notes:
|
||||
|
||||
00:00 Teaching machine learning, Python to graduate students.\
|
||||
06:03 Azure AI search service simplifies teaching, Qdrant facilitates learning.\
|
||||
10:49 Controversy over high jump style.\
|
||||
13:18 Embracing past for inspiration, emphasizing consistency.\
|
||||
15:43 Consistent learning and practice lead to success.\
|
||||
20:26 Teaching SQL uses SQLite, Rust has limitations.\
|
||||
25:21 Online platforms improve and speed up education.\
|
||||
29:24 Duke and Coursera offer specialized language courses.\
|
||||
31:21 Passion for wines, creating diverse dataset.\
|
||||
35:00 Encouragement for vector db discussion, wrap up.\
|
||||
|
||||
## More Quotes from Alfredo:
|
||||
|
||||
*"Qdrant makes it straightforward. We use it in-memory for my classes and I would love to see something similar setup in Rust to make teaching even easier.”*\
|
||||
— Alfredo Deza
|
||||
|
||||
*"Retrieval augmented generation is kind of like having an open book test. So the large language model is the student, and they have an open book so they can see the answers and then repackage that into their own words and provide an answer.”*\
|
||||
— Alfredo Deza
|
||||
|
||||
*"With Qdrant, I appreciate that the use of the Python API is so simple. It avoids the complexity that comes from having a back-end system like in Rust where you need an actual instance of the database running.”*\
|
||||
— Alfredo Deza
|
||||
|
||||
## Transcript:
|
||||
Demetrios:
|
||||
What is happening? Everyone, welcome back to another vector space talks. I am Demetrios, and I am joined today by good old Sabrina. Where you at, Sabrina? Hello?
|
||||
|
||||
Sabrina Aquino:
|
||||
Hello, Demetrios. I'm from Brazil. I'm in Brazil right now. I know that you are traveling currently.
|
||||
|
||||
Demetrios:
|
||||
Where are you? At Kubecon in Paris. And it has been magnificent. But I could not wait to join the session today because we've got Alfredo coming at us.
|
||||
|
||||
Alfredo Deza:
|
||||
What's up, dude? Hi. How are you?
|
||||
|
||||
Demetrios:
|
||||
I'm good, man. It's been a while. I think the last time that we chatted was two years ago, maybe right before your book came out. When did the book come out?
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, something like that. I would say a couple of years ago. Yeah. I wrote, co authored practical machine learning operations with no gift. And it was published on O'Reilly.
|
||||
|
||||
Demetrios:
|
||||
Yeah. And that was, I think, two years ago. So you've been doing a lot of stuff since then. Let's be honest, you are maybe one of the most active men on the Internet. I always love seeing what you're doing. You're bringing immense value to everything that you touch. I'm really excited to be able to chat with you for this next 30 minutes.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, of course.
|
||||
|
||||
Demetrios:
|
||||
Maybe just, we'll start it off. We're going to get into it when it comes to what you're doing and really what the space looks like right now. Right. But I would love to hear a little bit of what you've been up to since, for the last two years, because I haven't talked to you.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, that's right. Well, several different things, actually. Right after we chatted last time, I joined Microsoft to work in developer relations. Microsoft has a big group of folks working in developer relations. And basically, for me, it signaled my shift away from regular software engineering. I was primarily doing software engineering and thought that perhaps with the books and some of the courses that I had published, it was time for me to get into more teaching and providing useful content, which is really something very rewarding. And in developer relations, in advocacy in general, it's kind of like a way of teaching. We demonstrate technology, how it works from a technical point of view.
|
||||
|
||||
Alfredo Deza:
|
||||
So aside from that, started working really closely with several different universities. I work with Georgia Tech, Oxford University, Carnegie Mellon University, and Duke University, where I've been working as an adjunct professor for a couple of years as well. So at Duke, what I do is I teach a couple of classes a year. One is on machine learning. Last year was machine learning operations, and this year it's going to, I think, hopefully I'm not messing anything up. I think we're going to shift a little bit to doing operations with large language models. And in the fall I teach a programming class for graduate students that want to join one of the graduate programs and they want to get a primer on Python. So I teach a little bit of that.
|
||||
|
||||
Alfredo Deza:
|
||||
And in the meantime, also in partnership with Duke, getting a lot of courses out on Coursera, and from large language models to doing stuff with Azure, to machine learning operations, to rust, I've been doing a lot of rust lately, which I really like. So, yeah, so a lot of different things, but I think the core pillar for me remains being able to teach and spread the knowledge.
|
||||
|
||||
Demetrios:
|
||||
Love it, man. And I know you've been diving into vector databases. Can you tell us more?
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, well, the thing is that when you're trying to teach, and yes, one of the courses that we had out for large language models was applying retrieval augmented generation, which is the basis for vector databases, to see how it works. This is how it works. These are the components that you need. Let's create an application from scratch and see how it works. And for those that don't know, retrieval augmented generation is kind of like having. The other day I saw a description about this, which I really like, which is a way of, it's kind of like having an open book test. So the large language model is the student, and they have an open book so they can see the answers and then repackage that into their own words and provide an answer, which is kind of like what we do with vector databases in the retrieval augmented generation pattern. We've been putting a lot of examples on how to do these, and in the case of Azure, you're enabling certain services.
|
||||
|
||||
Alfredo Deza:
|
||||
There's the Azure AI search service, which is really good. But sometimes when you're trying to teach specifically, it is useful to have a very straightforward way to do this and applying or creating a retrieval augmented generation pattern, it's kind of tricky, I think. We're not there yet to do it in a nice, straightforward way. So there are several different options, Qdrant being one of them. So usually I get asked, why are you using Qdrant? What's the big deal? Why are you picking these over all of the other ones? And to me it boils down to, aside from being renowned or recognized, that it works fairly well. There's one core component that is critical here, and that is it has to be very straightforward, very easy to set up so that I can teach it, because if it's easy, well, sort of like easy to or straightforward to teach, then you can take the next step and you can make it a little more complex, put other things around it, and that creates a great development experience and a learning experience as well. If something is very complex, if the list of requirements is very long, you're not going to be very happy, you're going to spend all this time trying to figure, and when you have, similar to what happens with automation, when you have a list of 20 different things that you need to, in order to, say, deploy a website, you're going to get things out of order, you're going to forget one thing, you're going to have a typo, you're going to mess it up, you're going to have to start from scratch, and you're going to get into a situation where you can't get out of it. And Qdrant does provide a very straightforward way to run the database, and that one is the in memory implementation with Python.
|
||||
|
||||
Alfredo Deza:
|
||||
So you can actually write a little bit of python once you install the libraries and say, I want to instantiate a vector database and I wanted to run it in memory. So for teaching, this is great. It's like, hey, of course it's not for production, but just write these couple of lines and let's get right into it. Let's just start populating these and see how it works. And it works. It's great. You don't need to have all of these, like, wow, let's launch Kubernetes over here and let's have all of these dynamic. No, why? I mean, sure, you want to create a business model and you want to launch to production eventually, and you want to have all that running perfect.
|
||||
|
||||
Alfredo Deza:
|
||||
But for this setup, like for understanding how it works, for trying baby steps into understanding vector databases, this is perfect. My one requirement, or my one wish list item is to have that in memory thing for rust. That would be pretty sweet, because I think it'll make teaching rust and retrieval augmented generation with rust much easier. I wouldn't have to worry about bringing up containers or external services. So that's the deal with rust. And I'll tell you one last story about why I think specifically making it easy to get started with so that I can teach it, so that others can learn from it, is crucial. I would say almost 50 years ago, maybe a little bit more, my dad went to Italy to have a course on athletics. My dad was involved in sports and he was going through this, I think it was like a six month specialization on athletics.
|
||||
|
||||
Alfredo Deza:
|
||||
And he was in class and it had been recent that the high jump had transitioned from one style to the other. The previous style, the old style right now is the old style. It's kind of like, it was kind of like over the bar. It was kind of like a weird style. And it had recently transitioned to a thing called the Fosbury flop. This person, his last name is Dick Fosbury, invented the Fosbury flop. He said, no, I'm just going to go straight at it, then do a little curve and then jump over it. And then he did, and then he started winning everything.
|
||||
|
||||
Alfredo Deza:
|
||||
And everybody's like, what this guy? Well, first they thought he was crazy, and they thought that dismissive of what he was trying to do. And there were people that sticklers that wanted to stay with the older style, but then he started beating records and winning medals, and so people were like, well, is this a good thing? Let's try it out. So there was a whole. They were casting doubt. It's like, is this really the thing? Is this really what we should be doing? So one of the questions that my dad had to answer in this specialization he did in Italy was like, which style is better, it's the old style or the new style? And so my dad said, it's the new style. And they asked him, why is the new style better? And he didn't choose the path of answering the, well, because this guy just won the Olympics or he just did a record over here that at the end is meaningless. What he said was, it is the better style because it's easier to teach and it is 100% correct. When you're teaching high jump, it is much easier to teach the Fosbury flop than the other style.
|
||||
|
||||
Alfredo Deza:
|
||||
It is super hard. So you start seeing this parallel in teaching and learning where, but with this one, you have all of these world records and things are going great. Well, great. But is anybody going to try, are you going to have more people looking into it or are you going to have less? What is it that we're trying to do here? Right.
|
||||
|
||||
Demetrios:
|
||||
Not going to lie, I did not see how you were going to land the plane on coming from the high jump into the vector database space, but you did it gracefully. That was well done. So, basically, the easier it is to teach, the more people are going to be able to jump on board and the more people are going to be able to get value out of it.
|
||||
|
||||
Sabrina Aquino:
|
||||
I absolutely love it, by the way. It's a pleasure to meet you, Alfredo. And I was actually about to ask you. I love your background as an olympic athlete. Right. And I was wondering, do you make any connections or how do we interact this background with your current teaching and AI? And do you see any similarities or something coming from that approach into what you've applied?
|
||||
|
||||
Alfredo Deza:
|
||||
Well, you're bringing a great point. It's taken me a very long time to feel comfortable talking about my professional sports past. I don't want to feel like I'm overwhelming anyone or trying to be like a show off. So I usually try not to mention, although I'm feeling more comfortable mentioning my professional past. But the only situations where I think it's good to talk about it is when I feel like there's a small chance that I might get someone thinking about the possibilities of what they can actually do and what they can try. And things that are seemingly complex might be achievable. So you mentioned similarities, but I think there are a couple of things that happen when you're an athlete in any sport, really, that you're trying to or you're operating at the very highest level and there's several things that happen there. You have to be consistent.
|
||||
|
||||
Alfredo Deza:
|
||||
And it's something that I teach my kids as well. I have one of my kids, he's like, I did really a lot of exercise today and then for a week he doesn't do anything else. And he's like, now I'm going to do exercise again. And she's going to do 4 hours. And it's like, wait a second, wait a second. It's okay. You want to do it. This is great.
|
||||
|
||||
Alfredo Deza:
|
||||
But no intensity. You need to be consistent. Oh, dad, you don't let me work out and it's like, no work out. Good, I support you, but you have to be consistent and slowly start ramping up and slowly start getting better. And it happens a lot with learning. We are in an era that concepts and things are advancing so fast that things are getting obsolete even faster. So you're always in this motion of trying to learn. So what I would say is the similarities are in the consistency.
|
||||
|
||||
Alfredo Deza:
|
||||
You have to keep learning, you have to keep applying yourself. But it can be like, oh, today I'm going to read this whole book from start to end and you're just going to learn everything about, I don't know, rust. It's like, well, no, try applying rust a little bit every day and feel comfortable with it. And at the very end you will do better. Like, you can't go with high intensity because you're going to get burned out, you're going to overwhelmed and it's not going to work out. You don't go to the Olympics by working out for like a few months. Actually, a very long time ago, a reporter asked me, how many months have you been working out preparing for the Olympics? It's like, what do you mean with how many months? I've been training my whole life for this. What are we talking about?
|
||||
|
||||
Demetrios:
|
||||
We're not talking in months or years. We're talking in lifetimes, right?
|
||||
|
||||
Alfredo Deza:
|
||||
So you have to take it easy. You can't do that. And beyond that, consistency. Consistency goes hand in hand with discipline. I came to the US in 2006. I don't live like I was born in Peru and I came to the US with no degree. I didn't go to college. Well, I went to college for a few months and then I dropped out and I didn't have a career, I didn't have experience.
|
||||
|
||||
Alfredo Deza:
|
||||
I was just recently married. I have never worked in my life because I used to be a professional athlete. And the only thing that I decided to do was to do amazing work, apply myself and try to keep learning and never stop learning. In the back of my mind, it's like, oh, I have a tremendous knowledge gap that I need to fulfill by learning. And actually, I have tremendous respect and I'm incredibly grateful by all of the people that opened doors for me and gave me an opportunity, one of them being Noah Giff, which I co authored a few books with him and some of the courses. And he actually taught me to write Python. I didn't know how to program. And he said, you know what? I think you should learn to write some python.
|
||||
|
||||
Alfredo Deza:
|
||||
And I was like, python? Why would I ever need to do that? And I did. He's like, let's just find something to automate. I mean, what a concept. Find something to apply automation. And every week on Fridays, we'll just take a look at it and that's it. And we did that for a while. And then he said, you know what? You should apply for speaking at Python. How can I be speaking at a conference when I just started learning? It's like your perspective is different.
|
||||
|
||||
Alfredo Deza:
|
||||
You just started learning these. You're going to do it in an interesting way. So I think those are concepts that are very important to me. Stay disciplined, stay consistent, and keep at it. The secret is that there's no secret. That's the bottom line. You have to keep consistent. Otherwise things are always making excuses.
|
||||
|
||||
Alfredo Deza:
|
||||
Is very simple.
|
||||
|
||||
Demetrios:
|
||||
The secret is there is no secret. That is beautiful. So you did kind of sprinkle this idea of, oh, I wish there was more stuff happening with Qdrant and rust. Can you talk a little bit more to that? Because one piece of Qdrant that people tend to love is that it's built in rust. Right. But also, I know that you mentioned before, could we get a little bit of this action so that I don't have to deal with any. What was it you were saying? The containers.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah. Right. Now, if you want to have a proof of concept, and I always go for like, what's the easiest, the most straightforward, the less annoying things I need to do, the better. And with Python, the Python API for Qdrant, you can just write a few lines and say, I want to create an instance in memory and then that's it. The database is created for you. This is very similar, or I would say actually almost identical to how you run SQLite. Sqlite is the embedded database you can create in memory. And it's actually how I teach SQL as well.
|
||||
|
||||
Alfredo Deza:
|
||||
When I have to teach SQl, I use sqlite. I think it's perfect. But in rust, like you said, Qdrant's backend is built on rust. There is no in memory implementation. So you are required to have an actual instance of the Qdrant database running. So you have a couple of options, but one of them probably means you'll have to bring up a container with Qdrant running and then you'll have to connect to that instance. So when you're teaching, the development environments are kind of constrained. Either you are in a lab somewhere like Crusader has labs, but those are self contained.
|
||||
|
||||
Alfredo Deza:
|
||||
It's kind of tricky to get them running 100%. You can run multiple containers at the same time. So things start becoming more complex. Not only more complex for the learner, but also in this case, like the teacher, me who wants to figure out how to make this all run in a very constrained environment. And that makes it tricky. And I fasted the team, by the way, and I was told that maybe at some point they can do some magic and put the in memory implementation on the rust side of things, which I think it would be tremendous.
|
||||
|
||||
Sabrina Aquino:
|
||||
We're going to advocate for that on our side. We're also going to be asking for it. And I think this is really good too. It really makes it easier. Me as a student not long ago, I do see what you mean. It's quite hard to get it all working very fast in the time of a class that you don't have a lot of time and students can get. I don't know, it's quite complex. I do get what you mean.
|
||||
|
||||
Sabrina Aquino:
|
||||
And you also are working both on the tech industry and on academia, which I think is super interesting. And I always kind of feel like those two are a bit disconnected sometimes. And I was wondering what you think that how important is the collaboration of these two areas considering how fast the AI space is going to right now? And what are your thoughts?
|
||||
|
||||
Alfredo Deza:
|
||||
Well, I don't like generalizing, but I'm going to generalize right now. I would say most universities are several steps behind, and there's a lot of complexities involved in higher education specifically. Most importantly, these institutions tend to be fairly large, and with fairly large institutions, what do you get? Oh, you get the magical bureaucracy for anything you want to do. Something like, oh, well, you need to talk to that department that needs to authorize something, that needs to go to some other department, and it's like, I'm going to change the curriculum. It's like, no, you can't. What does that mean? I have actually had conversations with faculty in universities where they say, listen, curricula. Yeah, we get that. We need to update it, but we change curricula every five years.
|
||||
|
||||
Alfredo Deza:
|
||||
And so. See you in a while. It's been three years. We have two more years to go. See you in a couple of years. And that's detrimental to students now. I get it. Building curricula, it's very hard.
|
||||
|
||||
Alfredo Deza:
|
||||
It takes a lot of work for the faculty to put something together. So it is something that, from a faculty perspective, it's like they're not going to get paid more if they update the curriculum.
|
||||
|
||||
Demetrios:
|
||||
Right.
|
||||
|
||||
Alfredo Deza:
|
||||
And it's a massive amount of work now that, of course, comes to the detriment of the learner. The student will be under service because they will have to go through curricula that is fairly dated. Now, there are situations and there are programs where this doesn't happen. And Duke, I've worked with several. They're teaching Llama file, which was built by Mozilla. And when did Llama file came out? It was just like a few months ago. And I think it's incredible. And I think those skills that are the ones that students need today in order to not only learn these things, but also be able to apply them when they're looking for a job or trying to professionally even apply them into their day to day, now that's one side of things.
|
||||
|
||||
Alfredo Deza:
|
||||
But there's the other aspect. In the case of Duke, as well as other universities out there, they're using these online platforms so that they can put courses out there faster. Do you really need to go through a four year program to understand how retrieval augmented generation works? Or how to implement it? I would argue no, but would you be better out, like, taking a course that will take you perhaps a couple of weeks to go through and be fairly proficient? I would say yes, 100%. And you see several institutions putting courses out there that are meaningful, that are useful, that they can cope with the speed at which things are needed. I think it's kind of good. And I think that sometimes we tend to think about knowledge and learning things, kind of like in a bubble, especially here in the US. I think there's this college is this magical place where all of the amazing things happen. And if you don't go to college, things are going to go very bad for you.
|
||||
|
||||
Alfredo Deza:
|
||||
And I don't think that's true. I think if you like college, if you like university, by all means take advantage of it. You want to experience it. That sounds great. I think there's tons of opportunity to do it outside of the university or the college setting and taking online courses from validated instructors. They have a good profile. Not someone that just dumped something on genetic AI and started.
|
||||
|
||||
Demetrios:
|
||||
Someone like you.
|
||||
|
||||
Alfredo Deza:
|
||||
Well, if you want to. Yeah, sure, why not? I mean, there's students that really like my teaching style. I think that's great. If you don't like my teaching style. Sometimes I tend to go a little bit slower because I don't want to overwhelm anyone. That's all good. But there is opportunity. And when I mention these things, people are like, oh, really? I'm not advertising for Coursera or anything else, but some of these platforms, if you pay a monthly fee, I think it's between $40 and $60.
|
||||
|
||||
Alfredo Deza:
|
||||
I think on the expensive side, you can take advantage of all of these courses and as much as you can take them. Sometimes even companies say, hey, you have a paid subscription, go take it all. And I've met people like that. It's like, this is incredible. I'm learning so much. Perfect. I think there's a mix of things. I don't think there's like a binary answer, like, oh, you need to do this, or, no, don't do that, and everything's going to be well again.
|
||||
|
||||
Demetrios:
|
||||
Yeah. Can you talk a little bit more about your course? And if I wanted to go on Coursera, what can I expect from.
|
||||
|
||||
Alfredo Deza:
|
||||
You know, and again, I don't think as much as I like talking about my courses and the things that I do, I want to emphasize, like, if someone is watching this video or listening into what we're talking about, find something that is interesting to you and find a course that kind of delivers that thing, that sliver of interesting stuff, and then try it out. I think that's the best way. Don't get overwhelmed by. It's like, is this the right vector database that I should be learning? Is this instructor? It's like, no, try it out. What's going to happen? You don't like it when you're watching a bad video series or docuseries on Netflix or any streaming platform? Do you just like, I pay my $10 a month, so I'm going to muster through this whole 20 more episodes of this thing that I don't like. It's meaningless. It doesn't matter. Just move on.
|
||||
|
||||
Alfredo Deza:
|
||||
So having said that, on Coursera specifically with Duke University, we tend to put courses out there that are going to be used in our programs in the things that I teach. For example, we just released the large language models. Specialization and specialization is a grouping of between four and six courses. So in there we have doing large language models with Azure, for example, introduction to generative AI, having a very simple rag pattern with Qdrant. I also have examples on how to do it with Azure AI search, which I think is pretty cool as well. How to do it locally with Llama file, which I think is great. You can have all of these large language models running locally, and then you have a little bit of Qdrant sprinkle over there, and then you have rack pattern. Now, I tend to teach with things that I really like, and I'll give you a quick example.
|
||||
|
||||
Alfredo Deza:
|
||||
I think there's three data sets that are one of the top three most used data sets in all of machine learning and data science. Those are the Boston housing market, the diabetes data set in the US, and the other one is the Titanic. And everybody uses those. And I don't really understand why. I mean, perhaps I do understand why. It's because they're easy, they're clean, they're ready to go. Nothing's ever wrong with these, and everybody has used them to boredom. But for the life of me, you wouldn't be able to convince me to use any of those, because these are not topics that I really care about and they don't resonate with me.
|
||||
|
||||
Alfredo Deza:
|
||||
The Titanic specifically is just horrid. Well, if I was 37 and I'm on first class and I'm male, would I survive? It's like, what are we trying to do here? How is this useful to anyone? So I tend to use things that I like, and I'm really passionate about wine. So I built my own data set, which is a collection of wines from all over the world, they have the ratings, they have the region, they have the type of grape and the notes and the name of the wine. So when I'm teaching them, like, look at this, this is amazing. It's wines from all over the world. So let's do a little bit of things here. So, for rag, what I was able to do is actually in the courses as well. I do, ah, I really know wines from Argentina, but these wines, it would be amazing if you can find me not a Malbec, but perhaps a cabernet franc.
|
||||
|
||||
Alfredo Deza:
|
||||
That is amazing. From, it goes through Qdrant, goes back to llama file using some large language model or even small language model, like the Phi 2 from Microsoft, I think is really good. And he goes, it tells. Yeah, sure. I get that you want to have some good wines. Here's some good stuff that I can give you. And so it's great, right? I think it's great. So I think those kinds of things that are interesting to the person that is teaching or presenting, I think that's the key, because whenever you're talking about things that are very boring, that you do not care about, things are not going to go well for you.
|
||||
|
||||
Alfredo Deza:
|
||||
I mean, if I didn't like teaching, if I didn't like vector databases, you would tell right away. It's like, well, yes, I've been doing stuff with the vector databases. They're good. Yeah, Qdrant, very good. You would tell right away. I can't lie. Very good.
|
||||
|
||||
Demetrios:
|
||||
You can't fool anybody.
|
||||
|
||||
Alfredo Deza:
|
||||
No.
|
||||
|
||||
Demetrios:
|
||||
Well, dude, this is awesome. We will drop a link to the chat. We will drop a link to the course in the chat so that in case anybody does want to go on this wine tasting journey with you, they can. And I'm sure there's all kinds of things that will spark the creativity of the students as they go through it, because when you were talking about that, I was like, oh, it would be really cool to make that same type of thing, but with ski resorts there, you go around the world. And if I want this type of ski resort, I'm going to just ask my chat bot. So I'm excited to see what people create with it. I also really appreciate you coming on here, giving us your time and talking through all this. It's been a pleasure, as always, Alfredo.
|
||||
|
||||
Demetrios:
|
||||
Thank you so much.
|
||||
|
||||
Alfredo Deza:
|
||||
Yeah, thank you. Thank you for having me. Always happy to chat with you. I think Qdrant is doing a very solid product. Hopefully, my wish list item of in memory in rust comes to fruition, but I get it. Sometimes there are other priorities. It's all good. Yeah.
|
||||
|
||||
Alfredo Deza:
|
||||
If anyone wants to connect with me, I'm always active on LinkedIn primarily. Always happy to connect with folks and talk about learning and improving and always being a better person.
|
||||
|
||||
Demetrios:
|
||||
Excellent. Well, we will sign off, and if anyone else out there wants to come on here and talk to us about vector databases, we're always happy to have you. Feel free to reach out. And remember, don't get lost in vector space, folks. We will see you on the next one.
|
||||
|
||||
Sabrina Aquino:
|
||||
Good night. Thank you so much.
|
||||
@@ -63,10 +63,9 @@ curl -X PUT http://localhost:6333/collections/test_collection1 \
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -195,9 +194,8 @@ curl -X PUT http://localhost:6333/collections/test_collection2 \
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -325,10 +323,10 @@ curl -X PUT http://localhost:6333/collections/test_collection3 \
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -487,10 +485,9 @@ curl -X PUT http://localhost:6333/collections/test_collection4 \
|
||||
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -1419,7 +1416,7 @@ curl -X GET http://localhost:6333/collections/test_collection2/aliases
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.get_collection_aliases(collection_name="{collection_name}")
|
||||
```
|
||||
@@ -1472,7 +1469,7 @@ curl -X GET http://localhost:6333/aliases
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.get_aliases()
|
||||
```
|
||||
@@ -1525,7 +1522,7 @@ curl -X GET http://localhost:6333/collections
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.get_collections()
|
||||
```
|
||||
|
||||
@@ -36,10 +36,9 @@ POST /collections/{collection_name}/points/recommend
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.recommend(
|
||||
collection_name="{collection_name}",
|
||||
@@ -455,12 +454,11 @@ POST /collections/{collection_name}/points/recommend/batch
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
filter = models.Filter(
|
||||
filter_ = models.Filter(
|
||||
must=[
|
||||
models.FieldCondition(
|
||||
key="city",
|
||||
@@ -473,9 +471,9 @@ filter = models.Filter(
|
||||
|
||||
recommend_queries = [
|
||||
models.RecommendRequest(
|
||||
positive=[100, 231], negative=[718], filter=filter, limit=3
|
||||
positive=[100, 231], negative=[718], filter=filter_, limit=3
|
||||
),
|
||||
models.RecommendRequest(positive=[200, 67], negative=[300], filter=filter, limit=3),
|
||||
models.RecommendRequest(positive=[200, 67], negative=[300], filter=filter_, limit=3),
|
||||
]
|
||||
|
||||
client.recommend_batch(collection_name="{collection_name}", requests=recommend_queries)
|
||||
@@ -703,10 +701,9 @@ POST /collections/{collection_name}/points/discover
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
discover_queries = [
|
||||
models.DiscoverRequest(
|
||||
@@ -906,10 +903,9 @@ POST /collections/{collection_name}/points/discover
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
discover_queries = [
|
||||
models.DiscoverRequest(
|
||||
|
||||
@@ -52,10 +52,9 @@ POST /collections/{collection_name}/points/scroll
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient(host="localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.scroll(
|
||||
collection_name="{collection_name}",
|
||||
@@ -729,7 +728,7 @@ Example:
|
||||
```
|
||||
|
||||
```python
|
||||
FieldCondition(
|
||||
models.FieldCondition(
|
||||
key="color",
|
||||
match=models.MatchAny(any=["black", "yellow"]),
|
||||
)
|
||||
@@ -785,7 +784,7 @@ Example:
|
||||
```
|
||||
|
||||
```python
|
||||
FieldCondition(
|
||||
models.FieldCondition(
|
||||
key="color",
|
||||
match=models.MatchExcept(**{"except": ["black", "yellow"]}),
|
||||
)
|
||||
|
||||
@@ -36,7 +36,7 @@ PUT /collections/{collection_name}/index
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient(host="localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_payload_index(
|
||||
collection_name="{collection_name}",
|
||||
@@ -141,10 +141,9 @@ PUT /collections/{collection_name}/index
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient(host="localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_payload_index(
|
||||
collection_name="{collection_name}",
|
||||
@@ -316,16 +315,15 @@ PUT /collections/{collection_name}/index
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient(host="localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_payload_index(
|
||||
collection_name="{collection_name}",
|
||||
field_name="name_of_the_field_to_index",
|
||||
field_schema=models.IntegerIndexParams(
|
||||
type="integer",
|
||||
type=models.IntegerIndexType.INTEGER,
|
||||
lookup=False,
|
||||
range=True,
|
||||
),
|
||||
|
||||
@@ -200,10 +200,9 @@ PUT /collections/{collection_name}/points
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient(host="localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.upsert(
|
||||
collection_name="{collection_name}",
|
||||
@@ -610,6 +609,49 @@ await client.SetPayloadAsync(
|
||||
);
|
||||
```
|
||||
|
||||
_Available as of v1.8.0_
|
||||
|
||||
It is possible to modify only a specific key of the payload by using the `key` parameter.
|
||||
|
||||
For instance, given the following payload JSON object on a point:
|
||||
|
||||
```json
|
||||
{
|
||||
"property1": {
|
||||
"nested_property": "foo",
|
||||
},
|
||||
"property2": {
|
||||
"nested_property": "bar",
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
You can modify the `nested_property` of `property1` with the following request:
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/payload
|
||||
{
|
||||
"payload": {
|
||||
"nested_property": "qux",
|
||||
},
|
||||
"key": "property1",
|
||||
"points": [1]
|
||||
}
|
||||
```
|
||||
|
||||
Resulting in the following payload:
|
||||
|
||||
```json
|
||||
{
|
||||
"property1": {
|
||||
"nested_property": "qux",
|
||||
},
|
||||
"property2": {
|
||||
"nested_property": "bar",
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Overwrite payload
|
||||
|
||||
Fully replace any existing payload with the given one.
|
||||
@@ -725,9 +767,7 @@ POST /collections/{collection_name}/points/payload/clear
|
||||
```python
|
||||
client.clear_payload(
|
||||
collection_name="{collection_name}",
|
||||
points_selector=models.PointIdsList(
|
||||
points=[0, 3, 100],
|
||||
),
|
||||
points_selector=[0, 3, 100],
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
@@ -82,10 +82,9 @@ PUT /collections/{collection_name}/points
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.upsert(
|
||||
collection_name="{collection_name}",
|
||||
@@ -1212,9 +1211,7 @@ POST /collections/{collection_name}/points/vectors/delete
|
||||
```python
|
||||
client.delete_vectors(
|
||||
collection_name="{collection_name}",
|
||||
points_selector=models.PointIdsList(
|
||||
points=[0, 3, 100],
|
||||
),
|
||||
points=[0, 3, 100],
|
||||
vectors=["text", "image"],
|
||||
)
|
||||
```
|
||||
@@ -1955,7 +1952,7 @@ POST /collections/{collection_name}/points/batch
|
||||
|
||||
```python
|
||||
client.batch_update_points(
|
||||
collection_name=collection_name,
|
||||
collection_name="{collection_name}",
|
||||
update_operations=[
|
||||
models.UpsertOperation(
|
||||
upsert=models.PointsList(
|
||||
|
||||
@@ -83,10 +83,9 @@ POST /collections/{collection_name}/points/search
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -250,9 +249,8 @@ POST /collections/{collection_name}/points/search
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -358,10 +356,9 @@ POST /collections/{collection_name}/points/search
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -562,9 +559,8 @@ POST /collections/{collection_name}/points/search
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -656,10 +652,9 @@ POST /collections/{collection_name}/points/search
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -808,12 +803,11 @@ POST /collections/{collection_name}/points/search/batch
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
filter = models.Filter(
|
||||
filter_ = models.Filter(
|
||||
must=[
|
||||
models.FieldCondition(
|
||||
key="city",
|
||||
@@ -825,8 +819,8 @@ filter = models.Filter(
|
||||
)
|
||||
|
||||
search_queries = [
|
||||
models.SearchRequest(vector=[0.2, 0.1, 0.9, 0.7], filter=filter, limit=3),
|
||||
models.SearchRequest(vector=[0.5, 0.3, 0.2, 0.3], filter=filter, limit=3),
|
||||
models.SearchRequest(vector=[0.2, 0.1, 0.9, 0.7], filter=filter_, limit=3),
|
||||
models.SearchRequest(vector=[0.5, 0.3, 0.2, 0.3], filter=filter_, limit=3),
|
||||
]
|
||||
|
||||
client.search_batch(collection_name="{collection_name}", requests=search_queries)
|
||||
@@ -1003,7 +997,7 @@ POST /collections/{collection_name}/points/search
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -1185,7 +1179,7 @@ POST /collections/{collection_name}/points/search/groups
|
||||
client.search_groups(
|
||||
collection_name="{collection_name}",
|
||||
# Same as in the regular search() API
|
||||
query_vector=g,
|
||||
query_vector=[1.1],
|
||||
# Grouping parameters
|
||||
group_by="document_id", # Path of the field to group by
|
||||
limit=4, # Max amount of groups
|
||||
|
||||
@@ -51,7 +51,7 @@ POST /collections/{collection_name}/snapshots
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_snapshot(collection_name="{collection_name}")
|
||||
```
|
||||
@@ -103,7 +103,7 @@ DELETE /collections/{collection_name}/snapshots/{snapshot_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.delete_snapshot(
|
||||
collection_name="{collection_name}", snapshot_name="{snapshot_name}"
|
||||
@@ -155,7 +155,7 @@ GET /collections/{collection_name}/snapshots
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.list_snapshots(collection_name="{collection_name}")
|
||||
```
|
||||
@@ -241,7 +241,7 @@ PUT /collections/{collection_name}/snapshots/recover
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("qdrant-node-2", port=6333)
|
||||
client = QdrantClient(url="http://qdrant-node-2:6333")
|
||||
|
||||
client.recover_snapshot(
|
||||
"{collection_name}",
|
||||
@@ -326,7 +326,7 @@ PUT /collections/{collection_name}/snapshots/recover
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("qdrant-node-2", port=6333)
|
||||
client = QdrantClient(url="http://qdrant-node-2:6333")
|
||||
|
||||
client.recover_snapshot(
|
||||
"{collection_name}",
|
||||
@@ -371,7 +371,7 @@ POST /snapshots
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_full_snapshot()
|
||||
```
|
||||
@@ -421,7 +421,7 @@ DELETE /snapshots/{snapshot_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.delete_full_snapshot(snapshot_name="{snapshot_name}")
|
||||
```
|
||||
|
||||
@@ -56,7 +56,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -168,7 +168,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -293,7 +293,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
|
||||
@@ -17,6 +17,7 @@ be done in the following way:
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
|
||||
from aleph_alpha_client import (
|
||||
Prompt,
|
||||
@@ -25,7 +26,6 @@ from aleph_alpha_client import (
|
||||
SemanticRepresentation,
|
||||
ImagePrompt
|
||||
)
|
||||
from qdrant_client.http.models import Batch
|
||||
|
||||
aa_token = "<< your_token >>"
|
||||
model = "luminous-base"
|
||||
|
||||
@@ -35,7 +35,7 @@ bedrock_client = session.client(
|
||||
aws_secret_access_key="<YOUR_AWS_SECRET_KEY>",
|
||||
)
|
||||
|
||||
qdrant_client = QdrantClient(location="http://localhost:6333")
|
||||
qdrant_client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
qdrant_client.create_collection(
|
||||
"{collection_name}",
|
||||
|
||||
@@ -18,8 +18,7 @@ The embeddings returned by co.embed API might be used directly in the Qdrant cli
|
||||
```python
|
||||
import cohere
|
||||
import qdrant_client
|
||||
|
||||
from qdrant_client.http.models import Batch
|
||||
from qdrant_client.models import Batch
|
||||
|
||||
cohere_client = cohere.Client("<< your_api_key >>")
|
||||
qdrant_client = qdrant_client.QdrantClient()
|
||||
@@ -55,12 +54,11 @@ documents with the Embed v3 model:
|
||||
```python
|
||||
import cohere
|
||||
import qdrant_client
|
||||
|
||||
from qdrant_client.http.models import Batch
|
||||
from qdrant_client.models import Batch
|
||||
|
||||
cohere_client = cohere.Client("<< your_api_key >>")
|
||||
qdrant_client = qdrant_client.QdrantClient()
|
||||
qdrant_client.upsert(
|
||||
client = qdrant_client.QdrantClient()
|
||||
client.upsert(
|
||||
collection_name="MyCollection",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
@@ -76,9 +74,9 @@ qdrant_client.upsert(
|
||||
Once the documents are indexed, you can search for the most relevant documents using the Embed v3 model:
|
||||
|
||||
```python
|
||||
qdrant_client.search(
|
||||
client.search(
|
||||
collection_name="MyCollection",
|
||||
query=cohere_client.embed(
|
||||
query_vector=cohere_client.embed(
|
||||
model="embed-english-v3.0", # New Embed v3 model
|
||||
input_type="search_query", # Input type for search queries
|
||||
texts=["The best vector database"],
|
||||
|
||||
@@ -43,11 +43,13 @@ The following example shows how to embed a document with the `models/embedding-0
|
||||
```python
|
||||
import google.generativeai as gemini_client
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http.models import Distance, PointStruct, VectorParams
|
||||
from qdrant_client.models import Distance, PointStruct, VectorParams
|
||||
|
||||
collection_name = "example_collection"
|
||||
|
||||
GEMINI_API_KEY = "YOUR GEMINI API KEY" # add your key here
|
||||
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
gemini_client.configure(api_key=GEMINI_API_KEY)
|
||||
texts = [
|
||||
"Qdrant is a vector database that is compatible with Gemini.",
|
||||
@@ -83,7 +85,7 @@ points = [
|
||||
### Create Collection
|
||||
|
||||
```python
|
||||
search_client.create_collection(collection_name, vectors_config=
|
||||
client.create_collection(collection_name, vectors_config=
|
||||
VectorParams(
|
||||
size=768,
|
||||
distance=Distance.COSINE,
|
||||
@@ -94,7 +96,7 @@ search_client.create_collection(collection_name, vectors_config=
|
||||
### Add these into the collection
|
||||
|
||||
```python
|
||||
search_client.upsert(collection_name, points)
|
||||
client.upsert(collection_name, points)
|
||||
```
|
||||
|
||||
## Searching for documents with Qdrant
|
||||
@@ -102,7 +104,7 @@ search_client.upsert(collection_name, points)
|
||||
Once the documents are indexed, you can search for the most relevant documents using the same model with the `retrieval_query` task type:
|
||||
|
||||
```python
|
||||
search_client.search(
|
||||
client.search(
|
||||
collection_name=collection_name,
|
||||
query_vector=gemini_client.embed_content(
|
||||
model="models/embedding-001",
|
||||
|
||||
@@ -16,8 +16,7 @@ To call their endpoint, all you need is an API key obtainable [here](https://jin
|
||||
import qdrant_client
|
||||
import requests
|
||||
|
||||
from qdrant_client.http.models import Distance, VectorParams
|
||||
from qdrant_client.http.models import Batch
|
||||
from qdrant_client.models import Distance, VectorParams, Batch
|
||||
|
||||
# Provide Jina API key and choose one of the available models.
|
||||
# You can get a free trial key here: https://jina.ai/embeddings/
|
||||
@@ -43,8 +42,8 @@ embeddings = [d["embedding"] for d in response.json()["data"]]
|
||||
|
||||
|
||||
# Index the embeddings into Qdrant
|
||||
qdrant_client = qdrant_client.QdrantClient(":memory:")
|
||||
qdrant_client.create_collection(
|
||||
client = qdrant_client.QdrantClient(":memory:")
|
||||
client.create_collection(
|
||||
collection_name="MyCollection",
|
||||
vectors_config=VectorParams(size=EMBEDDING_SIZE, distance=Distance.DOT),
|
||||
)
|
||||
|
||||
@@ -22,11 +22,12 @@ And then we set this up:
|
||||
```python
|
||||
from mistralai.client import MistralClient
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http.models import PointStruct, VectorParams, Distance
|
||||
from qdrant_client.models import PointStruct, VectorParams, Distance
|
||||
|
||||
collection_name = "example_collection"
|
||||
|
||||
MISTRAL_API_KEY = "your_mistral_api_key"
|
||||
search_client = QdrantClient(":memory:")
|
||||
client = QdrantClient(":memory:")
|
||||
mistral_client = MistralClient(api_key=MISTRAL_API_KEY)
|
||||
texts = [
|
||||
"Qdrant is the best vector search engine!",
|
||||
@@ -65,13 +66,12 @@ points = [
|
||||
## Create a collection and Insert the documents
|
||||
|
||||
```python
|
||||
search_client.create_collection(collection_name, vectors_config=
|
||||
VectorParams(
|
||||
client.create_collection(collection_name, vectors_config=VectorParams(
|
||||
size=1024,
|
||||
distance=Distance.COSINE,
|
||||
)
|
||||
)
|
||||
search_client.upsert(collection_name, points)
|
||||
client.upsert(collection_name, points)
|
||||
```
|
||||
|
||||
## Searching for documents with Qdrant
|
||||
@@ -79,7 +79,7 @@ search_client.upsert(collection_name, points)
|
||||
Once the documents are indexed, you can search for the most relevant documents using the same model with the `retrieval_query` task type:
|
||||
|
||||
```python
|
||||
search_client.search(
|
||||
client.search(
|
||||
collection_name=collection_name,
|
||||
query_vector=mistral_client.embeddings(
|
||||
model="mistral-embed", input=["What is the best to use for vector search scaling?"]
|
||||
|
||||
@@ -30,8 +30,8 @@ output = embed.text(
|
||||
task_type="search_document",
|
||||
)
|
||||
|
||||
qdrant_client = QdrantClient()
|
||||
qdrant_client.upsert(
|
||||
client = QdrantClient()
|
||||
client.upsert(
|
||||
collection_name="my-collection",
|
||||
points=models.Batch(
|
||||
ids=[1],
|
||||
@@ -44,14 +44,14 @@ qdrant_client.upsert(
|
||||
|
||||
```python
|
||||
from fastembed import TextEmbedding
|
||||
from qdrant_client import QdrantClient, models
|
||||
from client import QdrantClient, models
|
||||
|
||||
model = TextEmbedding("nomic-ai/nomic-embed-text-v1")
|
||||
|
||||
output = model.embed(["Qdrant is the best vector database!"])
|
||||
|
||||
qdrant_client = QdrantClient()
|
||||
qdrant_client.upsert(
|
||||
client = QdrantClient()
|
||||
client.upsert(
|
||||
collection_name="my-collection",
|
||||
points=models.Batch(
|
||||
ids=[1],
|
||||
@@ -71,7 +71,7 @@ output = embed.text(
|
||||
task_type="search_query",
|
||||
)
|
||||
|
||||
qdrant_client.search(
|
||||
client.search(
|
||||
collection_name="my-collection",
|
||||
query_vector=output["embeddings"][0],
|
||||
)
|
||||
@@ -82,7 +82,7 @@ qdrant_client.search(
|
||||
```python
|
||||
output = next(model.embed("What is the best vector database?"))
|
||||
|
||||
qdrant_client.search(
|
||||
client.search(
|
||||
collection_name="my-collection",
|
||||
query_vector=output.tolist(),
|
||||
)
|
||||
|
||||
@@ -21,7 +21,7 @@ NVIDIA_API_KEY = "<YOUR_API_KEY>"
|
||||
|
||||
nvidia_session = requests.Session()
|
||||
|
||||
qdrant_client = QdrantClient(":memory:")
|
||||
client = QdrantClient(":memory:")
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {NVIDIA_API_KEY}",
|
||||
@@ -89,7 +89,7 @@ let response_body = await response.json()
|
||||
### Converting the model outputs to Qdrant points
|
||||
|
||||
```python
|
||||
from qdrant_client.http.models import PointStruct
|
||||
from qdrant_client.models import PointStruct
|
||||
|
||||
points = [
|
||||
PointStruct(
|
||||
@@ -120,14 +120,14 @@ from qdrant_client.models import VectorParams, Distance
|
||||
|
||||
collection_name = "example_collection"
|
||||
|
||||
qdrant_client.create_collection(
|
||||
client.create_collection(
|
||||
collection_name,
|
||||
vectors_config=VectorParams(
|
||||
size=1024,
|
||||
distance=Distance.COSINE,
|
||||
),
|
||||
)
|
||||
qdrant_client.upsert(collection_name, points)
|
||||
client.upsert(collection_name, points)
|
||||
```
|
||||
|
||||
```typescript
|
||||
@@ -161,7 +161,7 @@ response_body = nvidia_session.post(
|
||||
NVIDIA_BASE_URL, headers=headers, json=payload
|
||||
).json()
|
||||
|
||||
qdrant_client.search(
|
||||
client.search(
|
||||
collection_name=collection_name,
|
||||
query_vector=response_body["data"][0]["embedding"],
|
||||
)
|
||||
|
||||
@@ -24,7 +24,7 @@ openai_client = openai.Client(
|
||||
api_key="<YOUR_API_KEY>"
|
||||
)
|
||||
|
||||
qdrant_client = qdrant_client.QdrantClient(":memory:")
|
||||
client = qdrant_client.QdrantClient(":memory:")
|
||||
|
||||
texts = [
|
||||
"Qdrant is the best vector search engine!",
|
||||
@@ -39,13 +39,13 @@ The following example shows how to embed a document with the `text-embedding-3-s
|
||||
```python
|
||||
embedding_model = "text-embedding-3-small"
|
||||
|
||||
result = openai_client.embeddings.create(input= texts, model=embedding_model)
|
||||
result = openai_client.embeddings.create(input=texts, model=embedding_model)
|
||||
```
|
||||
|
||||
### Converting the model outputs to Qdrant points
|
||||
|
||||
```python
|
||||
from qdrant_client.http.models import PointStruct
|
||||
from qdrant_client.models import PointStruct
|
||||
|
||||
points = [
|
||||
PointStruct(
|
||||
@@ -60,18 +60,18 @@ points = [
|
||||
### Creating a collection to insert the documents
|
||||
|
||||
```python
|
||||
from qdrant_client.http.models import VectorParams, Distance
|
||||
from qdrant_client.models import VectorParams, Distance
|
||||
|
||||
collection_name = "example_collection"
|
||||
|
||||
qdrant_client.create_collection(
|
||||
client.create_collection(
|
||||
collection_name,
|
||||
vectors_config=VectorParams(
|
||||
size=1536,
|
||||
distance=Distance.COSINE,
|
||||
),
|
||||
)
|
||||
qdrant_client.upsert(collection_name, points)
|
||||
client.upsert(collection_name, points)
|
||||
```
|
||||
|
||||
## Searching for documents with Qdrant
|
||||
@@ -79,7 +79,7 @@ qdrant_client.upsert(collection_name, points)
|
||||
Once the documents are indexed, you can search for the most relevant documents using the same model.
|
||||
|
||||
```python
|
||||
qdrant_client.search(
|
||||
client.search(
|
||||
collection_name=collection_name,
|
||||
query_vector=openai_client.embeddings.create(
|
||||
input=["What is the best to use for vector search scaling?"],
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
---
|
||||
title: Voyage AI
|
||||
weight: 1300
|
||||
---
|
||||
|
||||
# Voyage AI
|
||||
|
||||
Qdrant supports working with [Voyage AI](https://voyageai.com/) embeddings. The supported models' list can be found [here](https://docs.voyageai.com/docs/embeddings).
|
||||
|
||||
You can generate an API key from the [Voyage AI dashboard](<https://dash.voyageai.com/>) to authenticate the requests.
|
||||
|
||||
### Setting up the Qdrant and Voyage clients
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
import voyageai
|
||||
|
||||
VOYAGE_API_KEY = "<YOUR_VOYAGEAI_API_KEY>"
|
||||
|
||||
qclient = QdrantClient(":memory:")
|
||||
vclient = voyageai.Client(api_key=VOYAGE_API_KEY)
|
||||
|
||||
texts = [
|
||||
"Qdrant is the best vector search engine!",
|
||||
"Loved by Enterprises and everyone building for low latency, high performance, and scale.",
|
||||
]
|
||||
```
|
||||
|
||||
```typescript
|
||||
import {QdrantClient} from '@qdrant/js-client-rest';
|
||||
|
||||
const VOYAGEAI_BASE_URL = "https://api.voyageai.com/v1/embeddings"
|
||||
const VOYAGEAI_API_KEY = "<YOUR_VOYAGEAI_API_KEY>"
|
||||
|
||||
const client = new QdrantClient({ url: 'http://localhost:6333' });
|
||||
|
||||
const headers = {
|
||||
"Authorization": "Bearer " + VOYAGEAI_API_KEY,
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
|
||||
const texts = [
|
||||
"Qdrant is the best vector search engine!",
|
||||
"Loved by Enterprises and everyone building for low latency, high performance, and scale.",
|
||||
]
|
||||
```
|
||||
|
||||
The following example shows how to embed documents with the [`voyage-large-2`](https://docs.voyageai.com/docs/embeddings#model-choices) model that generates sentence embeddings of size 1536.
|
||||
|
||||
### Embedding documents
|
||||
|
||||
```python
|
||||
response = vclient.embed(texts, model="voyage-large-2", input_type="document")
|
||||
```
|
||||
|
||||
```typescript
|
||||
let body = {
|
||||
"input": texts,
|
||||
"model": "voyage-large-2",
|
||||
"input_type": "document",
|
||||
}
|
||||
|
||||
let response = await fetch(VOYAGEAI_BASE_URL, {
|
||||
method: "POST",
|
||||
body: JSON.stringify(body),
|
||||
headers
|
||||
});
|
||||
|
||||
let response_body = await response.json();
|
||||
```
|
||||
|
||||
### Converting the model outputs to Qdrant points
|
||||
|
||||
```python
|
||||
from qdrant_client.models import PointStruct
|
||||
|
||||
points = [
|
||||
PointStruct(
|
||||
id=idx,
|
||||
vector=embedding,
|
||||
payload={"text": text},
|
||||
)
|
||||
for idx, (embedding, text) in enumerate(zip(response.embeddings, texts))
|
||||
]
|
||||
```
|
||||
|
||||
```typescript
|
||||
let points = response_body.data.map((data, i) => {
|
||||
return {
|
||||
id: i,
|
||||
vector: data.embedding,
|
||||
payload: {
|
||||
text: texts[i]
|
||||
}
|
||||
}
|
||||
});
|
||||
```
|
||||
|
||||
### Creating a collection to insert the documents
|
||||
|
||||
```python
|
||||
from qdrant_client.models import VectorParams, Distance
|
||||
|
||||
COLLECTION_NAME = "example_collection"
|
||||
|
||||
qclient.create_collection(
|
||||
COLLECTION_NAME,
|
||||
vectors_config=VectorParams(
|
||||
size=1536,
|
||||
distance=Distance.COSINE,
|
||||
),
|
||||
)
|
||||
qclient.upsert(COLLECTION_NAME, points)
|
||||
```
|
||||
|
||||
```typescript
|
||||
const COLLECTION_NAME = "example_collection"
|
||||
|
||||
await client.createCollection(COLLECTION_NAME, {
|
||||
vectors: {
|
||||
size: 1536,
|
||||
distance: 'Cosine',
|
||||
}
|
||||
});
|
||||
|
||||
await client.upsert(COLLECTION_NAME, {
|
||||
wait: true,
|
||||
points
|
||||
});
|
||||
```
|
||||
|
||||
### Searching for documents with Qdrant
|
||||
|
||||
Once the documents are added, you can search for the most relevant documents.
|
||||
|
||||
```python
|
||||
response = vclient.embed(
|
||||
["What is the best to use for vector search scaling?"],
|
||||
model="voyage-large-2",
|
||||
input_type="query",
|
||||
)
|
||||
|
||||
qclient.search(
|
||||
collection_name=COLLECTION_NAME,
|
||||
query_vector=response.embeddings[0],
|
||||
)
|
||||
```
|
||||
|
||||
```typescript
|
||||
body = {
|
||||
"input": ["What is the best to use for vector search scaling?"],
|
||||
"model": "voyage-large-2",
|
||||
"input_type": "query",
|
||||
};
|
||||
|
||||
response = await fetch(VOYAGEAI_BASE_URL, {
|
||||
method: "POST",
|
||||
body: JSON.stringify(body),
|
||||
headers
|
||||
});
|
||||
|
||||
response_body = await response.json();
|
||||
|
||||
await client.search(COLLECTION_NAME, {
|
||||
vector: response_body.data[0].embedding,
|
||||
});
|
||||
```
|
||||
@@ -69,7 +69,7 @@ ids = [32, 21, "b626f6a9-b14d-4af9-b7c3-43d8deb719a6"]
|
||||
payload = [{"meta": "data"}, {"meta": "data_2"}, {"meta": "data_3", "extra": "data"}]
|
||||
|
||||
QdrantIngestOperator(
|
||||
conn_id="qdrant_connection"
|
||||
conn_id="qdrant_connection",
|
||||
task_id="qdrant_ingest",
|
||||
collection_name="<COLLECTION_NAME>",
|
||||
vectors=vectors,
|
||||
|
||||
@@ -68,7 +68,7 @@ assistant = RetrieveAssistantAgent(
|
||||
# `chunk_token_size` is the chunk token size for the retrieve chat.
|
||||
# We use an in-memory QdrantClient instance here. Not recommended for production.
|
||||
|
||||
ragproxyagent = QdrantRetrieveUserProxyAgent(
|
||||
rag_proxy_agent = QdrantRetrieveUserProxyAgent(
|
||||
name="qdrantagent",
|
||||
human_input_mode="NEVER",
|
||||
max_consecutive_auto_reply=10,
|
||||
@@ -95,7 +95,7 @@ assistant.reset()
|
||||
|
||||
# The query used below is for demonstration. It should usually be related to the docs made available to the agent
|
||||
code_problem = "How can I use FLAML to perform a classification task?"
|
||||
ragproxyagent.initiate_chat(assistant, problem=code_problem)
|
||||
rag_proxy_agent.initiate_chat(assistant, problem=code_problem)
|
||||
```
|
||||
|
||||
## Next steps
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
---
|
||||
title: Pinecone Canopy
|
||||
weight: 2500
|
||||
---
|
||||
|
||||
# Pinecone Canopy
|
||||
|
||||
[Canopy](https://github.com/pinecone-io/canopy) is an open-source framework and context engine to build chat assistants at scale.
|
||||
|
||||
Qdrant is supported as a knowledge base within Canopy for context retrieval and augmented generation.
|
||||
|
||||
## Usage
|
||||
|
||||
Install the SDK with the Qdrant extra as described in the [Canopy README](https://github.com/pinecone-io/canopy?tab=readme-ov-file#extras).
|
||||
|
||||
```bash
|
||||
pip install canopy-sdk[qdrant]
|
||||
```
|
||||
|
||||
### Creating a knowledge base
|
||||
|
||||
```python
|
||||
from canopy.knowledge_base import QdrantKnowledgeBase
|
||||
|
||||
kb = QdrantKnowledgeBase(collection_name="<YOUR_COLLECTION_NAME>")
|
||||
```
|
||||
|
||||
<aside role="status">The constructor accepts additional <a href="https://github.com/qdrant/qdrant-client/blob/eda201a1dbf1bbc67415f8437a5619f6f83e8ac6/qdrant_client/qdrant_client.py#L36-L61">options</a> to customize your connection to Qdrant.</aside>
|
||||
|
||||
To create a new Qdrant collection and connect it to the knowledge base, use the `create_canopy_collection` method:
|
||||
|
||||
```python
|
||||
kb.create_canopy_collection()
|
||||
```
|
||||
|
||||
You can always verify the connection to the collection with the `verify_index_connection` method:
|
||||
|
||||
```python
|
||||
kb.verify_index_connection()
|
||||
```
|
||||
|
||||
Learn more about customizing the knowledge base and its inner components [in the Canopy library](https://github.com/pinecone-io/canopy/blob/main/docs/library.md#understanding-knowledgebase-workings).
|
||||
|
||||
### Adding data to the knowledge base
|
||||
|
||||
To insert data into the knowledge base, you can create a list of documents and use the `upsert` method:
|
||||
|
||||
```python
|
||||
from canopy.models.data_models import Document
|
||||
|
||||
documents = [
|
||||
Document(
|
||||
id="1",
|
||||
text="U2 are an Irish rock band from Dublin, formed in 1976.",
|
||||
source="https://en.wikipedia.org/wiki/U2",
|
||||
),
|
||||
Document(
|
||||
id="2",
|
||||
text="Arctic Monkeys are an English rock band formed in Sheffield in 2002.",
|
||||
source="https://en.wikipedia.org/wiki/Arctic_Monkeys",
|
||||
metadata={"my-key": "my-value"},
|
||||
),
|
||||
]
|
||||
|
||||
kb.upsert(documents)
|
||||
```
|
||||
|
||||
### Querying the knowledge base
|
||||
|
||||
You can query the knowledge base with the `query` method to find the most similar documents to a given text:
|
||||
|
||||
```python
|
||||
from canopy.models.data_models import Query
|
||||
|
||||
kb.query(
|
||||
[
|
||||
Query(text="Arctic Monkeys music genre"),
|
||||
Query(
|
||||
text="U2 music genre",
|
||||
top_k=10,
|
||||
metadata_filter={"key": "my-key", "match": {"value": "my-value"}},
|
||||
),
|
||||
]
|
||||
)
|
||||
```
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [Introduction to Canopy](https://www.pinecone.io/blog/canopy-rag-framework/)
|
||||
- [Canopy library reference](https://github.com/pinecone-io/canopy/blob/main/docs/library.md)
|
||||
@@ -27,7 +27,7 @@ Scalar Quantization, you'd make that in the following way:
|
||||
|
||||
```python
|
||||
from qdrant_haystack.document_stores import QdrantDocumentStore
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import models
|
||||
|
||||
document_store = QdrantDocumentStore(
|
||||
":memory:",
|
||||
|
||||
@@ -68,7 +68,8 @@ client is destroyed - usually at the end of your script/notebook.
|
||||
|
||||
```python
|
||||
qdrant = Qdrant.from_documents(
|
||||
docs, embeddings,
|
||||
docs,
|
||||
embeddings,
|
||||
location=":memory:", # Local mode with in-memory storage only
|
||||
collection_name="my_documents",
|
||||
)
|
||||
@@ -80,7 +81,8 @@ Local mode, without using the Qdrant server, may also store your vectors on disk
|
||||
|
||||
```python
|
||||
qdrant = Qdrant.from_documents(
|
||||
docs, embeddings,
|
||||
docs,
|
||||
embeddings,
|
||||
path="/tmp/local_qdrant",
|
||||
collection_name="my_documents",
|
||||
)
|
||||
|
||||
@@ -36,5 +36,5 @@ index = VectorStoreIndex.from_vector_store(vector_store=vector_store)
|
||||
|
||||
```
|
||||
|
||||
The library [comes with a notebook](https://github.com/run-llama/llama_index/blob/main/docs/examples/vector_stores/QdrantIndexDemo.ipynb)
|
||||
The library [comes with a notebook](https://colab.research.google.com/github/run-llama/llama_index/blob/main/docs/docs/examples/vector_stores/QdrantIndexDemo.ipynb)
|
||||
that shows an end-to-end example of how to use Qdrant within LlamaIndex.
|
||||
|
||||
@@ -15,7 +15,7 @@ pip install pandasai[qdrant]
|
||||
|
||||
## Usage
|
||||
|
||||
You can begin a conversation by instantiating an `Agent` instance based on your Pandas data frame. The default Pandas-AI LLM requires an [API key](https://pandabi.ai.).
|
||||
You can begin a conversation by instantiating an `Agent` instance based on your Pandas data frame. The default Pandas-AI LLM requires an [API key](https://pandabi.ai).
|
||||
|
||||
You can find the list of all supported LLMs [here](https://docs.pandas-ai.com/en/latest/LLMs/llms/)
|
||||
|
||||
@@ -62,7 +62,8 @@ from pandasai.ee.vectorstores.qdrant import Qdrant
|
||||
qdrant = Qdrant(
|
||||
collection_name="<SOME_COLLECTION>",
|
||||
embedding_model="sentence-transformers/all-MiniLM-L6-v2",
|
||||
location="http://localhost:6334",
|
||||
url="http://localhost:6333",
|
||||
grpc_port=6334,
|
||||
prefer_grpc=True
|
||||
)
|
||||
|
||||
|
||||
@@ -26,7 +26,6 @@ import (
|
||||
qdrantContainer, err := qdrant.RunContainer(ctx, testcontainers.WithImage("qdrant/qdrant"))
|
||||
```
|
||||
|
||||
<!--
|
||||
```typescript
|
||||
import { QdrantContainer } from "@testcontainers/qdrant";
|
||||
|
||||
@@ -38,7 +37,6 @@ from testcontainers.qdrant import QdrantContainer
|
||||
|
||||
qdrant_container = QdrantContainer("qdrant/qdrant").start()
|
||||
```
|
||||
-->
|
||||
|
||||
Testcontainers modules provide options/methods to configure ENVs, volumes, and virtually everything you can configure in a Docker container.
|
||||
|
||||
|
||||
@@ -38,7 +38,7 @@ unstructured-ingest \
|
||||
--verbose \
|
||||
qdrant \
|
||||
--collection-name "test" \
|
||||
--location "http://localhost:6333" \
|
||||
--url "http://localhost:6333" \
|
||||
--batch-size 80
|
||||
```
|
||||
|
||||
@@ -66,7 +66,7 @@ from unstructured.ingest.runner.writers.qdrant import QdrantWriter
|
||||
def get_writer() -> Writer:
|
||||
return QdrantWriter(
|
||||
connector_config=SimpleQdrantConfig(
|
||||
location="http://localhost:6333",
|
||||
url="http://localhost:6333",
|
||||
collection_name="test",
|
||||
),
|
||||
write_config=QdrantWriteConfig(batch_size=80),
|
||||
|
||||
@@ -186,10 +186,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -337,10 +336,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -445,10 +443,9 @@ PUT /collections/{collection_name}/points
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.upsert(
|
||||
collection_name="{collection_name}",
|
||||
@@ -662,10 +659,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -913,10 +909,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -1209,7 +1204,7 @@ client.upsert(
|
||||
[0.1, 0.1, 0.9],
|
||||
],
|
||||
),
|
||||
ordering="strong",
|
||||
ordering=models.WriteOrdering.STRONG,
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
@@ -221,7 +221,7 @@ POST /collections/{collection_name}/points/search
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -342,7 +342,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
|
||||
@@ -42,10 +42,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -196,10 +195,9 @@ POST /collections/{collection_name}/points/search
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -316,7 +314,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -467,10 +465,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -620,7 +617,7 @@ POST /collections/{collection_name}/points/search
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -734,7 +731,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -852,7 +849,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
|
||||
@@ -167,10 +167,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -335,10 +334,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -475,10 +473,9 @@ PUT /collections/{collection_name}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -630,10 +627,9 @@ POST /collections/{collection_name}/points/search
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.http import models
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -784,7 +780,7 @@ POST /collections/{collection_name}/points/search
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -923,7 +919,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -1077,7 +1073,7 @@ POST /collections/{collection_name}/points/search
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.search(
|
||||
collection_name="{collection_name}",
|
||||
@@ -1200,7 +1196,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
|
||||
@@ -63,8 +63,7 @@ curl \
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient(
|
||||
url="https://localhost",
|
||||
port=6333,
|
||||
url="https://localhost:6333",
|
||||
api_key="your_secret_api_key_here",
|
||||
)
|
||||
```
|
||||
@@ -186,8 +185,7 @@ curl -X GET https://localhost:6333
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient(
|
||||
url="https://localhost",
|
||||
port=6333,
|
||||
url="https://localhost:6333",
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
@@ -39,7 +39,7 @@ Qdrant is now accessible:
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
```
|
||||
|
||||
```typescript
|
||||
@@ -78,7 +78,7 @@ var client = new QdrantClient("localhost", 6334);
|
||||
You will be storing all of your vector data in a Qdrant collection. Let's call it `test_collection`. This collection will be using a dot product distance metric to compare vectors.
|
||||
|
||||
```python
|
||||
from qdrant_client.http.models import Distance, VectorParams
|
||||
from qdrant_client.models import Distance, VectorParams
|
||||
|
||||
client.create_collection(
|
||||
collection_name="test_collection",
|
||||
@@ -135,7 +135,7 @@ await client.CreateCollectionAsync(
|
||||
Let's now add a few vectors with a payload. Payloads are other data you want to associate with the vector:
|
||||
|
||||
```python
|
||||
from qdrant_client.http.models import PointStruct
|
||||
from qdrant_client.models import PointStruct
|
||||
|
||||
operation_info = client.upsert(
|
||||
collection_name="test_collection",
|
||||
@@ -524,7 +524,7 @@ See [payload and vector in the result](../concepts/search/#payload-and-vector-in
|
||||
We can narrow down the results further by filtering by payload. Let's find the closest results that include "London".
|
||||
|
||||
```python
|
||||
from qdrant_client.http.models import Filter, FieldCondition, MatchValue
|
||||
from qdrant_client.models import Filter, FieldCondition, MatchValue
|
||||
|
||||
search_result = client.search(
|
||||
collection_name="test_collection",
|
||||
|
||||
@@ -70,7 +70,7 @@ from aleph_alpha_client import (
|
||||
from glob import glob
|
||||
|
||||
ids, vectors, payloads = [], [], []
|
||||
async with AsyncClient(token=aa_token) as client:
|
||||
async with AsyncClient(token=aa_token) as aa_client:
|
||||
for i, image_path in enumerate(glob("./val2017/*.jpg")):
|
||||
# Convert the JPEG file into the embedding by calling
|
||||
# Aleph Alpha API
|
||||
@@ -82,7 +82,7 @@ async with AsyncClient(token=aa_token) as client:
|
||||
"compress_to_size": 128,
|
||||
}
|
||||
query_request = SemanticEmbeddingRequest(**query_params)
|
||||
query_response = await client.semantic_embed(request=query_request, model=model)
|
||||
query_response = await aa_client.semantic_embed(request=query_request, model=model)
|
||||
|
||||
# Finally store the id, vector and the payload
|
||||
ids.append(i)
|
||||
@@ -96,17 +96,17 @@ Add all created embeddings, along with their ids and payloads into the `COCO` co
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.http.models import Batch, VectorParams, Distance
|
||||
from qdrant_client.models import Batch, VectorParams, Distance
|
||||
|
||||
qdrant_client = qdrant_client.QdrantClient()
|
||||
qdrant_client.recreate_collection(
|
||||
client = qdrant_client.QdrantClient()
|
||||
client.recreate_collection(
|
||||
collection_name="COCO",
|
||||
vectors_config=VectorParams(
|
||||
size=len(vectors[0]),
|
||||
distance=Distance.COSINE,
|
||||
),
|
||||
)
|
||||
qdrant_client.upsert(
|
||||
client.upsert(
|
||||
collection_name="COCO",
|
||||
points=Batch(
|
||||
ids=ids,
|
||||
@@ -126,7 +126,7 @@ text queries and reverse image search. Assume you want to find images similar to
|
||||
With the following code snippet create its vector embedding and then perform the lookup in Qdrant:
|
||||
|
||||
```python
|
||||
async with AsyncCliet(token=aa_token) as client:
|
||||
async with AsyncCliet(token=aa_token) as aa_client:
|
||||
prompt = ImagePrompt.from_file("query.jpg")
|
||||
prompt = Prompt.from_image(prompt)
|
||||
|
||||
@@ -136,9 +136,9 @@ async with AsyncCliet(token=aa_token) as client:
|
||||
"compress_to_size": 128,
|
||||
}
|
||||
query_request = SemanticEmbeddingRequest(**query_params)
|
||||
query_response = await client.semantic_embed(request=query_request, model=model)
|
||||
query_response = await aa_client.semantic_embed(request=query_request, model=model)
|
||||
|
||||
results = qdrant.search(
|
||||
results = client.search(
|
||||
collection_name="COCO",
|
||||
query_vector=query_response.embedding,
|
||||
limit=3,
|
||||
@@ -156,16 +156,16 @@ and Spanish. Your search is not only multimodal, but also multilingual, without
|
||||
```python
|
||||
text = "Surfing"
|
||||
|
||||
async with AsyncClient(token=aa_token) as client:
|
||||
async with AsyncClient(token=aa_token) as aa_client:
|
||||
query_params = {
|
||||
"prompt": Prompt.from_text(text),
|
||||
"representation": SemanticRepresentation.Symmetric,
|
||||
"compres_to_size": 128,
|
||||
}
|
||||
query_request = SemanticEmbeddingRequest(**query_params)
|
||||
query_response = await client.semantic_embed(request=query_request, model=model)
|
||||
query_response = await aa_client.semantic_embed(request=query_request, model=model)
|
||||
|
||||
results = qdrant.search(
|
||||
results = client.search(
|
||||
collection_name="COCO",
|
||||
query_vector=query_response.embedding,
|
||||
limit=3,
|
||||
|
||||
@@ -7,7 +7,7 @@ weight: 14
|
||||
|
||||
Asynchronous programming is being broadly adopted in the Python ecosystem. Tools such as FastAPI [have embraced this new
|
||||
paradigm](https://fastapi.tiangolo.com/async/), but it is also becoming a standard for ML models served as SaaS. For example, the Cohere SDK
|
||||
[provides an async client](https://cohere-sdk.readthedocs.io/en/latest/cohere.html#asyncclient) next to its synchronous counterpart.
|
||||
[provides an async client](https://github.com/cohere-ai/cohere-python/blob/856a4c3bd29e7a75fa66154b8ac9fcdf1e0745e0/src/cohere/client.py#L189) next to its synchronous counterpart.
|
||||
|
||||
Databases are often launched as separate services and are accessed via a network. All the interactions with them are IO-bound and can
|
||||
be performed asynchronously so as not to waste time actively waiting for a server response. In Python, this is achieved by
|
||||
|
||||
@@ -37,7 +37,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -78,7 +78,7 @@ PATCH /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.update_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -138,7 +138,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
|
||||
@@ -14,6 +14,13 @@ models can now speak to the external tools and extract meaningful data on their
|
||||
source and let the Cohere LLM know how to access it. Obviously, vector search goes well with LLMs, and enabling semantic
|
||||
search over your data is a typical case.
|
||||
|
||||
Cohere RAG has lots of interesting features, such as inline citations, which help you to refer to the specific parts of
|
||||
the documents used to generate the response.
|
||||
|
||||

|
||||
|
||||
*Source: https://docs.cohere.com/docs/retrieval-augmented-generation-rag*
|
||||
|
||||
The connectors have to implement a specific interface and expose the data source as HTTP REST API. Cohere documentation
|
||||
[describes a general process of creating a connector](https://docs.cohere.com/docs/creating-and-deploying-a-connector).
|
||||
This tutorial guides you step by step on building such a service around Qdrant.
|
||||
@@ -35,11 +42,11 @@ actions to perform.
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
qdrant_client = QdrantClient(
|
||||
client = QdrantClient(
|
||||
"https://my-cluster.cloud.qdrant.io:6333",
|
||||
api_key="my-api-key",
|
||||
)
|
||||
qdrant_client.create_collection(
|
||||
client.create_collection(
|
||||
collection_name="personal-notes",
|
||||
vectors_config=models.VectorParams(
|
||||
size=1024,
|
||||
@@ -113,7 +120,7 @@ response = cohere_client.embed(
|
||||
input_type="search_document",
|
||||
)
|
||||
|
||||
qdrant_client.upload_points(
|
||||
client.upload_points(
|
||||
collection_name="personal-notes",
|
||||
points=[
|
||||
models.PointStruct(
|
||||
@@ -176,7 +183,7 @@ from typing import Annotated
|
||||
|
||||
app = FastAPI()
|
||||
|
||||
def qdrant_client() -> QdrantClient:
|
||||
def client() -> QdrantClient:
|
||||
return QdrantClient(config.QDRANT_URL, api_key=config.QDRANT_API_KEY)
|
||||
|
||||
def cohere_client() -> cohere.Client:
|
||||
@@ -185,7 +192,7 @@ def cohere_client() -> cohere.Client:
|
||||
@app.post("/search")
|
||||
def search(
|
||||
query: SearchQuery,
|
||||
qdrant_client: Annotated[QdrantClient, Depends(qdrant_client)],
|
||||
client: Annotated[QdrantClient, Depends(client)],
|
||||
cohere_client: Annotated[cohere.Client, Depends(cohere_client)],
|
||||
) -> SearchResults:
|
||||
response = cohere_client.embed(
|
||||
@@ -193,7 +200,7 @@ def search(
|
||||
model="embed-multilingual-v3.0",
|
||||
input_type="search_query",
|
||||
)
|
||||
results = qdrant_client.search(
|
||||
results = client.search(
|
||||
collection_name="personal-notes",
|
||||
query_vector=response.embeddings[0],
|
||||
limit=2,
|
||||
@@ -212,6 +219,11 @@ Our app might be launched locally for the development purposes, given we have th
|
||||
uvicorn main:app
|
||||
```
|
||||
|
||||
FastAPI exposes an interactive documentation at `http://localhost:8000/docs`, where we can test our endpoint. The
|
||||
`/search` endpoint is available there.
|
||||
|
||||

|
||||
|
||||
We can interact with it and check the documents that will be returned for a specific query. For example, we want to know
|
||||
recall what we are supposed to do regarding the infrastructure for your projects.
|
||||
|
||||
|
||||
@@ -75,9 +75,9 @@ We used the streaming mode, so the dataset is not loaded into memory. Instead, w
|
||||
|
||||
```python
|
||||
for payload in dataset:
|
||||
id = payload.pop("id")
|
||||
id_ = payload.pop("id")
|
||||
vector = payload.pop("vector")
|
||||
print(id, vector, payload)
|
||||
print(id_, vector, payload)
|
||||
```
|
||||
|
||||
A single payload looks like this:
|
||||
@@ -114,10 +114,10 @@ Calculating the embeddings is usually a bottleneck of the vector search pipeline
|
||||
```python
|
||||
ids, vectors, payloads = [], [], []
|
||||
for payload in dataset:
|
||||
id = payload.pop("id")
|
||||
id_ = payload.pop("id")
|
||||
vector = payload.pop("vector")
|
||||
|
||||
ids.append(id)
|
||||
ids.append(id_)
|
||||
vectors.append(vector)
|
||||
payloads.append(payload)
|
||||
|
||||
|
||||
@@ -106,7 +106,7 @@ Now you need to write a script to upload all startup data and vectors into the s
|
||||
# Import client library
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
qdrant_client = QdrantClient("http://localhost:6333")
|
||||
client = QdrantClient("http://localhost:6333")
|
||||
```
|
||||
|
||||
3. Select model to encode your data.
|
||||
@@ -114,16 +114,16 @@ qdrant_client = QdrantClient("http://localhost:6333")
|
||||
You will be using a pre-trained model called `sentence-transformers/all-MiniLM-L6-v2`.
|
||||
|
||||
```python
|
||||
qdrant_client.set_model("sentence-transformers/all-MiniLM-L6-v2")
|
||||
client.set_model("sentence-transformers/all-MiniLM-L6-v2")
|
||||
```
|
||||
|
||||
|
||||
4. Related vectors need to be added to a collection. Create a new collection for your startup vectors.
|
||||
|
||||
```python
|
||||
qdrant_client.recreate_collection(
|
||||
client.recreate_collection(
|
||||
collection_name="startups",
|
||||
vectors_config=qdrant_client.get_fastembed_vector_params(),
|
||||
vectors_config=client.get_fastembed_vector_params(),
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
@@ -146,13 +146,13 @@ Now you need to write a script to upload all startup data and vectors into the s
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.models import VectorParams, Distance
|
||||
|
||||
qdrant_client = QdrantClient("http://localhost:6333")
|
||||
client = QdrantClient("http://localhost:6333")
|
||||
```
|
||||
|
||||
3. Related vectors need to be added to a collection. Create a new collection for your startup vectors.
|
||||
|
||||
```python
|
||||
qdrant_client.recreate_collection(
|
||||
client.recreate_collection(
|
||||
collection_name="startups",
|
||||
vectors_config=VectorParams(size=384, distance=Distance.COSINE),
|
||||
)
|
||||
@@ -186,7 +186,7 @@ vectors = np.load("./startup_vectors.npy")
|
||||
5. Upload the data
|
||||
|
||||
```python
|
||||
qdrant_client.upload_collection(
|
||||
client.upload_collection(
|
||||
collection_name="startups",
|
||||
vectors=vectors,
|
||||
payload=payload,
|
||||
|
||||
@@ -105,10 +105,10 @@ after receiving the response from the `upsert` endpoint. **As long as the indexi
|
||||
the exact search**. We have to wait until the indexing is finished to be sure that the approximate search is performed.
|
||||
|
||||
```python
|
||||
client.upload_records(
|
||||
client.upload_points( # upload_points is available as of qdrant-client v1.7.1
|
||||
collection_name="arxiv-titles-instructorxl-embeddings",
|
||||
records=[
|
||||
models.Record(
|
||||
points=[
|
||||
models.PointStruct(
|
||||
id=item["id"],
|
||||
vector=item["vector"],
|
||||
payload=item,
|
||||
|
||||
@@ -147,7 +147,7 @@ documents = [
|
||||
You need to tell Qdrant where to store embeddings. This is a basic demo, so your local computer will use its memory as temporary storage.
|
||||
|
||||
```python
|
||||
qdrant = QdrantClient(":memory:")
|
||||
client = QdrantClient(":memory:")
|
||||
```
|
||||
|
||||
## 4. Create a collection
|
||||
@@ -155,7 +155,7 @@ qdrant = QdrantClient(":memory:")
|
||||
All data in Qdrant is organized by collections. In this case, you are storing books, so we are calling it `my_books`.
|
||||
|
||||
```python
|
||||
qdrant.recreate_collection(
|
||||
client.recreate_collection(
|
||||
collection_name="my_books",
|
||||
vectors_config=models.VectorParams(
|
||||
size=encoder.get_sentence_embedding_dimension(), # Vector size is defined by used model
|
||||
@@ -176,7 +176,7 @@ qdrant.recreate_collection(
|
||||
Tell the database to upload `documents` to the `my_books` collection. This will give each record an id and a payload. The payload is just the metadata from the dataset.
|
||||
|
||||
```python
|
||||
qdrant.upload_points(
|
||||
client.upload_points(
|
||||
collection_name="my_books",
|
||||
points=[
|
||||
models.PointStruct(
|
||||
@@ -192,7 +192,7 @@ qdrant.upload_points(
|
||||
Now that the data is stored in Qdrant, you can ask it questions and receive semantically relevant results.
|
||||
|
||||
```python
|
||||
hits = qdrant.search(
|
||||
hits = client.search(
|
||||
collection_name="my_books",
|
||||
query_vector=encoder.encode("alien invasion").tolist(),
|
||||
limit=3,
|
||||
@@ -216,7 +216,7 @@ The search engine shows three of the most likely responses that have to do with
|
||||
How about the most recent book from the early 2000s?
|
||||
|
||||
```python
|
||||
hits = qdrant.search(
|
||||
hits = client.search(
|
||||
collection_name="my_books",
|
||||
query_vector=encoder.encode("alien invasion").tolist(),
|
||||
query_filter=models.Filter(
|
||||
|
||||
@@ -7,4 +7,3 @@ sitemapExclude: True
|
||||
Current advances in NLP can reduce the retinue work of customer service by up to 80 percent.
|
||||
No more answering the same questions over and over again. A chatbot will do that, and people can focus on complex problems.
|
||||
But not only automated answering, it is also possible to control the quality of the department and automatically identify flaws in conversations.
|
||||
Read more about the "[Sentence Embeddings for Customer Support](https://blog.floydhub.com/automate-customer-support-part-one/)" case study.
|
||||
|
After Width: | Height: | Size: 25 KiB |
|
After Width: | Height: | Size: 15 KiB |
|
After Width: | Height: | Size: 106 KiB |
|
After Width: | Height: | Size: 73 KiB |
|
After Width: | Height: | Size: 45 KiB |
|
After Width: | Height: | Size: 618 KiB |
|
After Width: | Height: | Size: 988 KiB |
|
After Width: | Height: | Size: 28 KiB |
|
After Width: | Height: | Size: 21 KiB |
|
After Width: | Height: | Size: 178 KiB |
|
After Width: | Height: | Size: 116 KiB |
|
After Width: | Height: | Size: 84 KiB |
|
After Width: | Height: | Size: 543 KiB |
|
After Width: | Height: | Size: 415 KiB |
|
After Width: | Height: | Size: 21 KiB |
|
After Width: | Height: | Size: 16 KiB |
|
After Width: | Height: | Size: 137 KiB |
|
After Width: | Height: | Size: 88 KiB |
|
After Width: | Height: | Size: 61 KiB |
|
After Width: | Height: | Size: 21 KiB |
|
After Width: | Height: | Size: 16 KiB |
|
After Width: | Height: | Size: 130 KiB |
|
After Width: | Height: | Size: 85 KiB |
|
After Width: | Height: | Size: 59 KiB |
|
After Width: | Height: | Size: 599 KiB |
|
After Width: | Height: | Size: 625 KiB |
|
After Width: | Height: | Size: 68 KiB |
|
After Width: | Height: | Size: 60 KiB |
@@ -14,101 +14,85 @@
|
||||
<meta name="keywords" content="{{ range .Site.Params.Keywords }}{{ . }}, {{ end }} qdrant">
|
||||
{{ end }}
|
||||
|
||||
{{ if .IsHome }}
|
||||
<script type="application/ld+json">
|
||||
{
|
||||
"@context": "https://schema.org",
|
||||
"@type": "Organization",
|
||||
"name": "Qdrant",
|
||||
"legalName" : "Qdrant Solutions GmbH",
|
||||
"url": "https://qdrant.tech",
|
||||
"email": "info@qdrant.com",
|
||||
"@id": "https://qdrant.tech",
|
||||
"logo": "https://qdrant.tech/images/logo_with_text.png",
|
||||
"headline" : "{{ .Title }}",
|
||||
"description" : "{{ .Site.Params.description }}",
|
||||
"keywords" : [ {{ range .Site.Params.Keywords }}"{{ . }}", {{ end }} "Qdrant" ],
|
||||
"foundingDate": "2021",
|
||||
"founders": [
|
||||
{
|
||||
"@type": "Person",
|
||||
"name": "{{ .Site.Params.Author }}"
|
||||
}, {
|
||||
"@type": "Person",
|
||||
"name": "Andre Zayarni"
|
||||
}
|
||||
],
|
||||
"location": "Berlin, Germany",
|
||||
"address": {
|
||||
"@type": "PostalAddress",
|
||||
"streetAddress": "Chausseestraße 86",
|
||||
"addressLocality": "Berlin",
|
||||
"addressRegion": "Berlin",
|
||||
"postalCode": "10115",
|
||||
"addressCountry": "DE"
|
||||
},
|
||||
"contactPoint": {
|
||||
"@type": "ContactPoint",
|
||||
"contactType": "customer support",
|
||||
"telephone": "+49 3040797694",
|
||||
"email": "info@qdrant.com"
|
||||
},
|
||||
"sameAs": [
|
||||
{{ .Site.Params.github }},
|
||||
{{ .Site.Params.twitter }},
|
||||
{{ .Site.Params.linkedin }},
|
||||
{{ .Site.Params.discord }}
|
||||
],
|
||||
"mainEntityOfPage": {
|
||||
"@type": "SoftwareApplication",
|
||||
"name": "Qdrant",
|
||||
"applicationCategory": "Vector Search Engine",
|
||||
"operatingSystem": "linux, macOS",
|
||||
"downloadUrl": "https://github.com/qdrant/qdrant",
|
||||
"installUrl": "https://hub.docker.com/r/qdrant/qdrant",
|
||||
"abstract": "{{ .Site.Params.description }}",
|
||||
"image": "https://qdrant.tech/images/logo_with_text.png"
|
||||
}
|
||||
}
|
||||
</script>
|
||||
{{ else if .Params.seo_schema }}
|
||||
<!-- todo: use as default option -->
|
||||
<script type="application/ld+json">
|
||||
{{ .Params.seo_schema }}
|
||||
</script>
|
||||
{{ else if and (eq .Section "articles") .IsPage }}
|
||||
{{ if .Params.seo_schema_json }} <!-- if schema is defined in front matter as a list of json files -->
|
||||
<script type="application/ld+json">
|
||||
{
|
||||
"@context": "https://schema.org",
|
||||
"@type": "Article",
|
||||
"headline": "{{ .Params.title }}",
|
||||
"image": [
|
||||
{{ .Params.social_preview_image | absURL }},
|
||||
],
|
||||
"abstract": {{ .Params.description }},
|
||||
"datePublished": {{ .Params.date }},
|
||||
"dateModified": {{ .Params.date }},
|
||||
"author": [{
|
||||
"@type": "Person",
|
||||
"name": {{ .Params.author }},
|
||||
"url": {{ .Params.author_link }}
|
||||
}]
|
||||
"@graph":
|
||||
[
|
||||
{{- $context := . -}}
|
||||
{{- range $index, $element := .Params.seo_schema_json -}}
|
||||
{{- $template := resources.Get $element -}}
|
||||
{{- $schema := $template | resources.ExecuteAsTemplate (printf "schema%d.json" $index) $context -}}
|
||||
{{ $schema | unmarshal }}
|
||||
{{- if not (eq $index (sub (len $.Params.seo_schema_json) 1)) -}},
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
]
|
||||
}
|
||||
</script>
|
||||
{{ else if and (eq .Section "case-studies") .IsPage }}
|
||||
|
||||
{{ else if .IsHome }}
|
||||
|
||||
{{ $organizationTemplate := resources.Get "schema/organization-schema.json" }}
|
||||
{{ $organizationSchema := $organizationTemplate | resources.ExecuteAsTemplate "schema-organization-home.json" . }}
|
||||
|
||||
{{ $productTemplate := resources.Get "schema/product-schema.json" }}
|
||||
{{ $productSchema := $productTemplate | resources.ExecuteAsTemplate "schema-product-home.json" . }}
|
||||
<script type="application/ld+json">
|
||||
{
|
||||
"@context": "https://schema.org",
|
||||
"@type": "Article",
|
||||
"headline": "{{ .Params.title }}",
|
||||
"image": [
|
||||
{{ .Params.social_preview_image | absURL }},
|
||||
],
|
||||
"abstract": {{ .Params.description }},
|
||||
"datePublished": {{ .Params.date }},
|
||||
"dateModified": {{ .Params.date }}
|
||||
"@graph": [
|
||||
{{ $organizationSchema | unmarshal }},
|
||||
{{ $productSchema | unmarshal }}
|
||||
]
|
||||
}
|
||||
</script>
|
||||
|
||||
{{ else if or (in (slice "documentation" "benchmarks") .Section) (and (in (slice "blog" "articles") .Section) .IsPage) }}
|
||||
|
||||
{{ $organizationTemplate := resources.Get "schema/organization-schema.json" }}
|
||||
{{ $organizationTarget := printf "schema-org-1-%s.json" (replaceRE "(\\s)" "" .Params.title) }}
|
||||
{{ $organizationSchema := $organizationTemplate | resources.ExecuteAsTemplate $organizationTarget . }}
|
||||
|
||||
{{ $articleTemplate := resources.Get "schema/article-schema.json" }}
|
||||
{{ $articleTarget := printf "schema-art-%s.json" (replaceRE "(\\s)" "" .Params.title) }}
|
||||
{{ $articleSchema := $articleTemplate | resources.ExecuteAsTemplate $articleTarget . }}
|
||||
<script type="application/ld+json">
|
||||
{
|
||||
"@context": "https://schema.org",
|
||||
"@graph": [
|
||||
{{ $articleSchema | unmarshal }},
|
||||
{{ $organizationSchema | unmarshal }}
|
||||
]
|
||||
}
|
||||
</script>
|
||||
|
||||
{{ else }}
|
||||
|
||||
{{ $organizationTemplate := resources.Get "schema/organization-schema.json" }}
|
||||
{{ $organizationTarget := printf "schema-org-2-%s.json" (replaceRE "(\\s)" "" .Params.title) }}
|
||||
{{ $organizationSchema := $organizationTemplate | resources.ExecuteAsTemplate $organizationTarget . }}
|
||||
|
||||
{{ $productTemplate := resources.Get "schema/product-schema.json" }}
|
||||
{{ $productTarget := printf "schema-prod-%s.json" (replaceRE "(\\s)" "" .Params.title) }}
|
||||
{{ $productSchema := $productTemplate | resources.ExecuteAsTemplate $productTarget . }}
|
||||
<script type="application/ld+json">
|
||||
{
|
||||
"@context": "https://schema.org",
|
||||
"@graph": [
|
||||
{{ $organizationSchema | unmarshal }},
|
||||
{{ $productSchema | unmarshal }}
|
||||
]
|
||||
}
|
||||
</script>
|
||||
|
||||
{{ end }}
|
||||
|
||||
{{ if .Params.seo_schema }} <!-- if schema is defined in front matter -->
|
||||
<script type="application/ld+json">
|
||||
{{ .Params.seo_schema | jsonify | unmarshal }}
|
||||
</script>
|
||||
{{ end }}
|
||||
|
||||
{{ $url := urls.Parse .Site.BaseURL }}
|
||||
|
||||