mirror of
https://github.com/langchain-ai/langchain.git
synced 2026-10-10 03:45:17 +03:00
10 KiB
10 KiB
In [1]:
from langchain.chains import VectorDBQA
from langchain.llms import OpenAIIn [2]:
from langchain.document_loaders import TextLoader
loader = TextLoader('../state_of_the_union.txt')In [3]:
from langchain.indexes import VectorstoreIndexCreatorIn [4]:
index = VectorstoreIndexCreator().from_loaders([loader])Running Chroma using direct local API. Using DuckDB in-memory for database. Data will be transient.
In [5]:
query = "What did the president say about Ketanji Brown Jackson"
index.query(query)Out [5]:
" The president said that Ketanji Brown Jackson is one of the nation's top legal minds, a former top litigator in private practice, a former federal public defender, and from a family of public school educators and police officers. He also said that she is a consensus builder and has received a broad range of support from the Fraternal Order of Police to former judges appointed by Democrats and Republicans."
In [6]:
query = "What did the president say about Ketanji Brown Jackson"
index.query_with_sources(query)Out [6]:
{'question': 'What did the president say about Ketanji Brown Jackson',
'answer': " The president said that he nominated Circuit Court of Appeals Judge Ketanji Brown Jackson, one of the nation's top legal minds, to continue Justice Breyer's legacy of excellence, and that she has received a broad range of support from the Fraternal Order of Police to former judges appointed by Democrats and Republicans.\n",
'sources': '../state_of_the_union.txt'}In [7]:
index.vectorstoreOut [7]:
<langchain.vectorstores.chroma.Chroma at 0x113a3a700>
In [6]:
documents = loader.load()In [8]:
from langchain.text_splitter import CharacterTextSplitter
text_splitter = CharacterTextSplitter(chunk_size=1000, chunk_overlap=0)
texts = text_splitter.split_documents(documents)In [10]:
from langchain.embeddings import OpenAIEmbeddings
embeddings = OpenAIEmbeddings()In [11]:
from langchain.vectorstores import Chroma
db = Chroma.from_documents(texts, embeddings)Running Chroma using direct local API. Using DuckDB in-memory for database. Data will be transient.
In [12]:
qa = VectorDBQA.from_chain_type(llm=OpenAI(), chain_type="stuff", vectorstore=db)In [13]:
query = "What did the president say about Ketanji Brown Jackson"
qa.run(query)Out [13]:
" The President said that Ketanji Brown Jackson is one of the nation's top legal minds and a consensus builder, with a broad range of support from the Fraternal Order of Police to former judges appointed by Democrats and Republicans. She is a former top litigator in private practice, a former federal public defender, and from a family of public school educators and police officers."
In [14]:
index_creator = VectorstoreIndexCreator(
vectorstore_cls=Chroma,
embedding=OpenAIEmbeddings(),
text_splitter=CharacterTextSplitter(chunk_size=1000, chunk_overlap=0)
)In [ ]: