mirror of
https://github.com/langchain-ai/langchain.git
synced 2026-10-09 19:35:20 +03:00
- **Description**: [BagelDB](bageldb.ai) a collaborative vector database. Integrated the bageldb PyPi package with langchain with related tests and code. - **Issue**: Not applicable. - **Dependencies**: `betabageldb` PyPi package. - **Tag maintainer**: @rlancemartin, @eyurtsev, @baskaryan - **Twitter handle**: bageldb_ai (https://twitter.com/BagelDB_ai) We ran `make format`, `make lint` and `make test` locally. Followed the contribution guideline thoroughly https://github.com/hwchase17/langchain/blob/master/.github/CONTRIBUTING.md --------- Co-authored-by: Towhid1 <nurulaktertowhid@gmail.com>
7.2 KiB
7.2 KiB
In [9]:
from langchain.vectorstores import Bagel
texts = ["hello bagel", "hello langchain", "I love salad", "my car", "a dog"]
# create cluster and add texts
cluster = Bagel.from_texts(cluster_name="testing", texts=texts)In [11]:
# similarity search
cluster.similarity_search("bagel", k=3)Out [11]:
[Document(page_content='hello bagel', metadata={}),
Document(page_content='my car', metadata={}),
Document(page_content='I love salad', metadata={})]In [12]:
# the score is a distance metric, so lower is better
cluster.similarity_search_with_score("bagel", k=3)Out [12]:
[(Document(page_content='hello bagel', metadata={}), 0.27392977476119995),
(Document(page_content='my car', metadata={}), 1.4783176183700562),
(Document(page_content='I love salad', metadata={}), 1.5342965126037598)]In [13]:
# delete the cluster
cluster.delete_cluster()In [33]:
from langchain.document_loaders import TextLoader
from langchain.text_splitter import CharacterTextSplitter
loader = TextLoader("../../../state_of_the_union.txt")
documents = loader.load()
text_splitter = CharacterTextSplitter(chunk_size=1000, chunk_overlap=0)
docs = text_splitter.split_documents(documents)[:10]In [36]:
# create cluster with docs
cluster = Bagel.from_documents(cluster_name="testing_with_docs", documents=docs)In [37]:
# similarity search
query = "What did the president say about Ketanji Brown Jackson"
docs = cluster.similarity_search(query)
print(docs[0].page_content[:102])Madam Speaker, Madam Vice President, our First Lady and Second Gentleman. Members of Congress and the
In [53]:
texts = ["hello bagel", "this is langchain"]
cluster = Bagel.from_texts(cluster_name="testing", texts=texts)
cluster_data = cluster.get()In [54]:
# all keys
cluster_data.keys()Out [54]:
dict_keys(['ids', 'embeddings', 'metadatas', 'documents'])
In [56]:
# all values and keys
cluster_dataOut [56]:
{'ids': ['578c6d24-3763-11ee-a8ab-b7b7b34f99ba',
'578c6d25-3763-11ee-a8ab-b7b7b34f99ba',
'fb2fc7d8-3762-11ee-a8ab-b7b7b34f99ba',
'fb2fc7d9-3762-11ee-a8ab-b7b7b34f99ba',
'6b40881a-3762-11ee-a8ab-b7b7b34f99ba',
'6b40881b-3762-11ee-a8ab-b7b7b34f99ba',
'581e691e-3762-11ee-a8ab-b7b7b34f99ba',
'581e691f-3762-11ee-a8ab-b7b7b34f99ba'],
'embeddings': None,
'metadatas': [{}, {}, {}, {}, {}, {}, {}, {}],
'documents': ['hello bagel',
'this is langchain',
'hello bagel',
'this is langchain',
'hello bagel',
'this is langchain',
'hello bagel',
'this is langchain']}In [57]:
cluster.delete_cluster()In [63]:
texts = ["hello bagel", "this is langchain"]
metadatas = [{"source": "notion"}, {"source": "google"}]
cluster = Bagel.from_texts(cluster_name="testing", texts=texts, metadatas=metadatas)
cluster.similarity_search_with_score("hello bagel", where={"source": "notion"})Out [63]:
[(Document(page_content='hello bagel', metadata={'source': 'notion'}), 0.0)]In [64]:
# delete the cluster
cluster.delete_cluster()