mirror of
https://github.com/langchain-ai/langchain.git
synced 2026-10-10 03:45:17 +03:00
45 KiB
45 KiB
In [ ]:
!pip install redis redisvl openai tiktokenIn [2]:
import os
import getpass
os.environ["OPENAI_API_KEY"] = getpass.getpass("OpenAI API Key:")In [3]:
from langchain.embeddings import OpenAIEmbeddings
embeddings = OpenAIEmbeddings()In [4]:
metadata = [
{
"user": "john",
"age": 18,
"job": "engineer",
"credit_score": "high",
},
{
"user": "derrick",
"age": 45,
"job": "doctor",
"credit_score": "low",
},
{
"user": "nancy",
"age": 94,
"job": "doctor",
"credit_score": "high",
},
{
"user": "tyler",
"age": 100,
"job": "engineer",
"credit_score": "high",
},
{
"user": "joe",
"age": 35,
"job": "dentist",
"credit_score": "medium",
},
]
texts = ["foo", "foo", "foo", "bar", "bar"]In [7]:
from langchain.vectorstores.redis import Redis
rds = Redis.from_texts(
texts,
embeddings,
metadatas=metadata,
redis_url="redis://localhost:6379",
index_name="users"
)In [6]:
rds.index_nameOut [6]:
'users'
In [7]:
# assumes you're running Redis locally (use --host, --port, --password, --username, to change this)
!rvl index listall[32m16:58:26[0m [34m[RedisVL][0m [1;30mINFO[0m Indices: [32m16:58:26[0m [34m[RedisVL][0m [1;30mINFO[0m 1. users
In [8]:
!rvl index info -i usersIndex Information: ╭──────────────┬────────────────┬───────────────┬─────────────────┬────────────╮ │ Index Name │ Storage Type │ Prefixes │ Index Options │ Indexing │ ├──────────────┼────────────────┼───────────────┼─────────────────┼────────────┤ │ users │ HASH │ ['doc:users'] │ [] │ 0 │ ╰──────────────┴────────────────┴───────────────┴─────────────────┴────────────╯ Index Fields: ╭────────────────┬────────────────┬─────────┬────────────────┬────────────────╮ │ Name │ Attribute │ Type │ Field Option │ Option Value │ ├────────────────┼────────────────┼─────────┼────────────────┼────────────────┤ │ user │ user │ TEXT │ WEIGHT │ 1 │ │ job │ job │ TEXT │ WEIGHT │ 1 │ │ credit_score │ credit_score │ TEXT │ WEIGHT │ 1 │ │ content │ content │ TEXT │ WEIGHT │ 1 │ │ age │ age │ NUMERIC │ │ │ │ content_vector │ content_vector │ VECTOR │ │ │ ╰────────────────┴────────────────┴─────────┴────────────────┴────────────────╯
In [9]:
!rvl stats -i usersStatistics: ╭─────────────────────────────┬─────────────╮ │ Stat Key │ Value │ ├─────────────────────────────┼─────────────┤ │ num_docs │ 5 │ │ num_terms │ 15 │ │ max_doc_id │ 5 │ │ num_records │ 33 │ │ percent_indexed │ 1 │ │ hash_indexing_failures │ 0 │ │ number_of_uses │ 4 │ │ bytes_per_record_avg │ 4.60606 │ │ doc_table_size_mb │ 0.000524521 │ │ inverted_sz_mb │ 0.000144958 │ │ key_table_size_mb │ 0.000193596 │ │ offset_bits_per_record_avg │ 8 │ │ offset_vectors_sz_mb │ 2.19345e-05 │ │ offsets_per_term_avg │ 0.69697 │ │ records_per_doc_avg │ 6.6 │ │ sortable_values_size_mb │ 0 │ │ total_indexing_time │ 0.32 │ │ total_inverted_index_blocks │ 16 │ │ vector_index_sz_mb │ 6.0126 │ ╰─────────────────────────────┴─────────────╯
In [10]:
results = rds.similarity_search("foo")
print(results[0].page_content)foo
In [11]:
# return metadata
results = rds.similarity_search("foo", k=3)
meta = results[1].metadata
print("Key of the document in Redis: ", meta.pop("id"))
print("Metadata of the document: ", meta)Key of the document in Redis: doc:users:a70ca43b3a4e4168bae57c78753a200f
Metadata of the document: {'user': 'derrick', 'job': 'doctor', 'credit_score': 'low', 'age': '45'}
In [12]:
# with scores (distances)
results = rds.similarity_search_with_score("foo", k=5)
for result in results:
print(f"Content: {result[0].page_content} --- Score: {result[1]}")Content: foo --- Score: 0.0 Content: foo --- Score: 0.0 Content: foo --- Score: 0.0 Content: bar --- Score: 0.1566 Content: bar --- Score: 0.1566
In [13]:
# limit the vector distance that can be returned
results = rds.similarity_search_with_score("foo", k=5, distance_threshold=0.1)
for result in results:
print(f"Content: {result[0].page_content} --- Score: {result[1]}")Content: foo --- Score: 0.0 Content: foo --- Score: 0.0 Content: foo --- Score: 0.0
In [14]:
# with scores
results = rds.similarity_search_with_relevance_scores("foo", k=5)
for result in results:
print(f"Content: {result[0].page_content} --- Similiarity: {result[1]}")Content: foo --- Similiarity: 1.0 Content: foo --- Similiarity: 1.0 Content: foo --- Similiarity: 1.0 Content: bar --- Similiarity: 0.8434 Content: bar --- Similiarity: 0.8434
In [15]:
# limit scores (similarities have to be over .9)
results = rds.similarity_search_with_relevance_scores("foo", k=5, score_threshold=0.9)
for result in results:
print(f"Content: {result[0].page_content} --- Similarity: {result[1]}")Content: foo --- Similarity: 1.0 Content: foo --- Similarity: 1.0 Content: foo --- Similarity: 1.0
In [16]:
# you can also add new documents as follows
new_document = ["baz"]
new_metadata = [{
"user": "sam",
"age": 50,
"job": "janitor",
"credit_score": "high"
}]
# both the document and metadata must be lists
rds.add_texts(new_document, new_metadata)Out [16]:
['doc:users:b9c71d62a0a34241a37950b448dafd38']
In [17]:
# now query the new document
results = rds.similarity_search("baz", k=3)
print(results[0].metadata){'id': 'doc:users:b9c71d62a0a34241a37950b448dafd38', 'user': 'sam', 'job': 'janitor', 'credit_score': 'high', 'age': '50'}
In [10]:
# use maximal marginal relevance search to diversify results
results = rds.max_marginal_relevance_search("foo")In [11]:
# the lambda_mult parameter controls the diversity of the results, the lower the more diverse
results = rds.max_marginal_relevance_search("foo", lambda_mult=0.1)In [18]:
# write the schema to a yaml file
rds.write_schema("redis_schema.yaml")In [19]:
# now we can connect to our existing index as follows
new_rds = Redis.from_existing_index(
embeddings,
index_name="users",
redis_url="redis://localhost:6379",
schema="redis_schema.yaml"
)
results = new_rds.similarity_search("foo", k=3)
print(results[0].metadata){'id': 'doc:users:8484c48a032d4c4cbe3cc2ed6845fabb', 'user': 'john', 'job': 'engineer', 'credit_score': 'high', 'age': '18'}
In [20]:
# see the schemas are the same
new_rds.schema == rds.schemaOut [20]:
True
In [21]:
# create a new index with the new schema defined above
index_schema = {
"tag": [{"name": "credit_score"}],
"text": [{"name": "user"}, {"name": "job"}],
"numeric": [{"name": "age"}],
}
rds, keys = Redis.from_texts_return_keys(
texts,
embeddings,
metadatas=metadata,
redis_url="redis://localhost:6379",
index_name="users_modified",
index_schema=index_schema, # pass in the new index schema
)
`index_schema` does not match generated metadata schema.
If you meant to manually override the schema, please ignore this message.
index_schema: {'tag': [{'name': 'credit_score'}], 'text': [{'name': 'user'}, {'name': 'job'}], 'numeric': [{'name': 'age'}]}
generated_schema: {'text': [{'name': 'user'}, {'name': 'job'}, {'name': 'credit_score'}], 'numeric': [{'name': 'age'}], 'tag': []}
In [22]:
from langchain.vectorstores.redis import RedisText
is_engineer = RedisText("job") == "engineer"
results = rds.similarity_search("foo", k=3, filter=is_engineer)
print("Job:", results[0].metadata["job"])
print("Engineers in the dataset:", len(results))Job: engineer Engineers in the dataset: 2
In [23]:
# fuzzy match
starts_with_doc = RedisText("job") % "doc*"
results = rds.similarity_search("foo", k=3, filter=starts_with_doc)
for result in results:
print("Job:", result.metadata["job"])
print("Jobs in dataset that start with 'doc':", len(results))Job: doctor Job: doctor Jobs in dataset that start with 'doc': 2
In [24]:
from langchain.vectorstores.redis import RedisNum
is_over_18 = RedisNum("age") > 18
is_under_99 = RedisNum("age") < 99
age_range = is_over_18 & is_under_99
results = rds.similarity_search("foo", filter=age_range)
for result in results:
print("User:", result.metadata["user"], "is", result.metadata["age"])User: derrick is 45 User: nancy is 94 User: joe is 35
In [25]:
# make sure to use parenthesis around FilterExpressions
# if initializing them while constructing them
age_range = (RedisNum("age") > 18) & (RedisNum("age") < 99)
results = rds.similarity_search("foo", filter=age_range)
for result in results:
print("User:", result.metadata["user"], "is", result.metadata["age"])User: derrick is 45 User: nancy is 94 User: joe is 35
In [26]:
query = "foo"
results = rds.similarity_search_with_score(query, k=3, return_metadata=True)
for result in results:
print("Content:", result[0].page_content, " --- Score: ", result[1])
Content: foo --- Score: 0.0 Content: foo --- Score: 0.0 Content: foo --- Score: 0.0
In [27]:
retriever = rds.as_retriever(search_type="similarity", search_kwargs={"k": 4})In [28]:
docs = retriever.get_relevant_documents(query)
docsOut [28]:
[Document(page_content='foo', metadata={'id': 'doc:users_modified:988ecca7574048e396756efc0e79aeca', 'user': 'john', 'job': 'engineer', 'credit_score': 'high', 'age': '18'}),
Document(page_content='foo', metadata={'id': 'doc:users_modified:009b1afeb4084cc6bdef858c7a99b48e', 'user': 'derrick', 'job': 'doctor', 'credit_score': 'low', 'age': '45'}),
Document(page_content='foo', metadata={'id': 'doc:users_modified:7087cee9be5b4eca93c30fbdd09a2731', 'user': 'nancy', 'job': 'doctor', 'credit_score': 'high', 'age': '94'}),
Document(page_content='bar', metadata={'id': 'doc:users_modified:01ef6caac12b42c28ad870aefe574253', 'user': 'tyler', 'job': 'engineer', 'credit_score': 'high', 'age': '100'})]In [29]:
retriever = rds.as_retriever(search_type="similarity_distance_threshold", search_kwargs={"k": 4, "distance_threshold": 0.1})In [30]:
docs = retriever.get_relevant_documents(query)
docsOut [30]:
[Document(page_content='foo', metadata={'id': 'doc:users_modified:988ecca7574048e396756efc0e79aeca', 'user': 'john', 'job': 'engineer', 'credit_score': 'high', 'age': '18'}),
Document(page_content='foo', metadata={'id': 'doc:users_modified:009b1afeb4084cc6bdef858c7a99b48e', 'user': 'derrick', 'job': 'doctor', 'credit_score': 'low', 'age': '45'}),
Document(page_content='foo', metadata={'id': 'doc:users_modified:7087cee9be5b4eca93c30fbdd09a2731', 'user': 'nancy', 'job': 'doctor', 'credit_score': 'high', 'age': '94'})]In [31]:
retriever = rds.as_retriever(search_type="similarity_score_threshold", search_kwargs={"score_threshold": 0.9, "k": 10})In [32]:
retriever.get_relevant_documents("foo")Out [32]:
[Document(page_content='foo', metadata={'id': 'doc:users_modified:988ecca7574048e396756efc0e79aeca', 'user': 'john', 'job': 'engineer', 'credit_score': 'high', 'age': '18'}),
Document(page_content='foo', metadata={'id': 'doc:users_modified:009b1afeb4084cc6bdef858c7a99b48e', 'user': 'derrick', 'job': 'doctor', 'credit_score': 'low', 'age': '45'}),
Document(page_content='foo', metadata={'id': 'doc:users_modified:7087cee9be5b4eca93c30fbdd09a2731', 'user': 'nancy', 'job': 'doctor', 'credit_score': 'high', 'age': '94'})]In [12]:
retriever = rds.as_retriever(search_type="mmr", search_kwargs={"fetch_k": 20, "k": 4, "lambda_mult": 0.1})In [13]:
retriever.get_relevant_documents("foo")Out [13]:
[Document(page_content='foo', metadata={'id': 'doc:users:8f6b673b390647809d510112cde01a27', 'user': 'john', 'job': 'engineer', 'credit_score': 'high', 'age': '18'}),
Document(page_content='bar', metadata={'id': 'doc:users:93521560735d42328b48c9c6f6418d6a', 'user': 'tyler', 'job': 'engineer', 'credit_score': 'high', 'age': '100'}),
Document(page_content='foo', metadata={'id': 'doc:users:125ecd39d07845eabf1a699d44134a5b', 'user': 'nancy', 'job': 'doctor', 'credit_score': 'high', 'age': '94'}),
Document(page_content='foo', metadata={'id': 'doc:users:d6200ab3764c466082fde3eaab972a2a', 'user': 'derrick', 'job': 'doctor', 'credit_score': 'low', 'age': '45'})]In [33]:
Redis.delete(keys, redis_url="redis://localhost:6379")Out [33]:
True
In [34]:
# delete the indices too
Redis.drop_index(index_name="users", delete_documents=True, redis_url="redis://localhost:6379")
Redis.drop_index(index_name="users_modified", delete_documents=True, redis_url="redis://localhost:6379")Out [34]:
True
In [35]:
# connection to redis standalone at localhost, db 0, no password
redis_url = "redis://localhost:6379"
# connection to host "redis" port 7379 with db 2 and password "secret" (old style authentication scheme without username / pre 6.x)
redis_url = "redis://:secret@redis:7379/2"
# connection to host redis on default port with user "joe", pass "secret" using redis version 6+ ACLs
redis_url = "redis://joe:secret@redis/0"
# connection to sentinel at localhost with default group mymaster and db 0, no password
redis_url = "redis+sentinel://localhost:26379"
# connection to sentinel at host redis with default port 26379 and user "joe" with password "secret" with default group mymaster and db 0
redis_url = "redis+sentinel://joe:secret@redis"
# connection to sentinel, no auth with sentinel monitoring group "zone-1" and database 2
redis_url = "redis+sentinel://redis:26379/zone-1/2"
# connection to redis standalone at localhost, db 0, no password but with TLS support
redis_url = "rediss://localhost:6379"
# connection to redis sentinel at localhost and default port, db 0, no password
# but with TLS support for booth Sentinel and Redis server
redis_url = "rediss+sentinel://localhost"