mirror of
https://github.com/langchain-ai/langchain.git
synced 2026-10-11 04:15:12 +03:00
8.7 KiB
8.7 KiB
In [1]:
# Comment this out if you are NOT using tracing
import os
os.environ["LANGCHAIN_HANDLER"] = "langchain"In [2]:
from langchain.evaluation.loading import load_dataset
dataset = load_dataset("question-answering-paul-graham")Found cached dataset json (/Users/harrisonchase/.cache/huggingface/datasets/LangChainDatasets___json/LangChainDatasets--question-answering-paul-graham-76e8f711e038d742/0.0.0/0f7e3662623656454fcd2b650f34e886a7db4b9104504885bd462096cc7a9f51)
0%| | 0/1 [00:00<?, ?it/s]
In [3]:
from langchain.document_loaders import TextLoader
loader = TextLoader("../../modules/paul_graham_essay.txt")In [4]:
from langchain.indexes import VectorstoreIndexCreatorIn [5]:
vectorstore = VectorstoreIndexCreator().from_loaders([loader]).vectorstoreRunning Chroma using direct local API. Using DuckDB in-memory for database. Data will be transient.
In [6]:
from langchain.chains import RetrievalQA
from langchain.llms import OpenAIIn [7]:
chain = RetrievalQA.from_chain_type(llm=OpenAI(), chain_type="stuff", retriever=vectorstore.as_retriever(), input_key="question")In [18]:
chain(dataset[0])Out [18]:
{'question': 'What were the two main things the author worked on before college?',
'answer': 'The two main things the author worked on before college were writing and programming.',
'result': ' Writing and programming.'}In [9]:
predictions = chain.apply(dataset)In [10]:
predictions[0]Out [10]:
{'question': 'What were the two main things the author worked on before college?',
'answer': 'The two main things the author worked on before college were writing and programming.',
'result': ' Writing and programming.'}In [11]:
from langchain.evaluation.qa import QAEvalChainIn [12]:
llm = OpenAI(temperature=0)
eval_chain = QAEvalChain.from_llm(llm)
graded_outputs = eval_chain.evaluate(dataset, predictions, question_key="question", prediction_key="result")In [13]:
for i, prediction in enumerate(predictions):
prediction['grade'] = graded_outputs[i]['text']In [14]:
from collections import Counter
Counter([pred['grade'] for pred in predictions])Out [14]:
Counter({' CORRECT': 12, ' INCORRECT': 10})In [15]:
incorrect = [pred for pred in predictions if pred['grade'] == " INCORRECT"]In [16]:
incorrect[0]Out [16]:
{'question': 'What did the author write their dissertation on?',
'answer': 'The author wrote their dissertation on applications of continuations.',
'result': ' The author does not mention what their dissertation was on, so it is not known.',
'grade': ' INCORRECT'}In [ ]: