mirror of
https://github.com/langchain-ai/langchain.git
synced 2026-10-09 11:25:21 +03:00
Still don't have good "how to's", and the guides / examples section could be further pruned and improved, but this PR adds a couple examples for each of the common evaluator interfaces. - [x] Example docs for each implemented evaluator - [x] "how to make a custom evalutor" notebook for each low level APIs (comparison, string, agent) - [x] Move docs to modules area - [x] Link to reference docs for more information - [X] Still need to finish the evaluation index page - ~[ ] Don't have good data generation section~ - ~[ ] Don't have good how to section for other common scenarios / FAQs like regression testing, testing over similar inputs to measure sensitivity, etc.~
5.2 KiB
5.2 KiB
In [1]:
# %pip install rapidfuzzIn [2]:
from langchain.evaluation import load_evaluator
evaluator = load_evaluator("string_distance")In [3]:
evaluator.evaluate_strings(
prediction="The job is completely done.",
reference="The job is done",
)Out [3]:
{'score': 12}In [4]:
# The results purely character-based, so it's less useful when negation is concerned
evaluator.evaluate_strings(
prediction="The job is done.",
reference="The job isn't done",
)Out [4]:
{'score': 4}In [5]:
from langchain.evaluation import StringDistance
list(StringDistance)Out [5]:
[<StringDistance.DAMERAU_LEVENSHTEIN: 'damerau_levenshtein'>, <StringDistance.LEVENSHTEIN: 'levenshtein'>, <StringDistance.JARO: 'jaro'>, <StringDistance.JARO_WINKLER: 'jaro_winkler'>]
In [6]:
jaro_evaluator = load_evaluator(
"string_distance", distance=StringDistance.JARO, requires_reference=True
)In [7]:
jaro_evaluator.evaluate_strings(
prediction="The job is completely done.",
reference="The job is done",
)Out [7]:
{'score': 0.19259259259259254}In [8]:
jaro_evaluator.evaluate_strings(
prediction="The job is done.",
reference="The job isn't done",
)Out [8]:
{'score': 0.12083333333333324}