mirror of
https://github.com/langchain-ai/langchain.git
synced 2026-10-09 11:25:21 +03:00
Still don't have good "how to's", and the guides / examples section could be further pruned and improved, but this PR adds a couple examples for each of the common evaluator interfaces. - [x] Example docs for each implemented evaluator - [x] "how to make a custom evalutor" notebook for each low level APIs (comparison, string, agent) - [x] Move docs to modules area - [x] Link to reference docs for more information - [X] Still need to finish the evaluation index page - ~[ ] Don't have good data generation section~ - ~[ ] Don't have good how to section for other common scenarios / FAQs like regression testing, testing over similar inputs to measure sensitivity, etc.~
5.4 KiB
5.4 KiB
In [1]:
# %pip install evaluate > /dev/nullIn [2]:
from typing import Any, Optional
from langchain.evaluation import StringEvaluator
from evaluate import load
class PerplexityEvaluator(StringEvaluator):
"""Evaluate the perplexity of a predicted string."""
def __init__(self, model_id: str = "gpt2"):
self.model_id = model_id
self.metric_fn = load(
"perplexity", module_type="metric", model_id=self.model_id, pad_token=0
)
def _evaluate_strings(
self,
*,
prediction: str,
reference: Optional[str] = None,
input: Optional[str] = None,
**kwargs: Any,
) -> dict:
results = self.metric_fn.compute(
predictions=[prediction], model_id=self.model_id
)
ppl = results["perplexities"][0]
return {"score": ppl}In [3]:
evaluator = PerplexityEvaluator()In [4]:
evaluator.evaluate_strings(prediction="The rains in Spain fall mainly on the plain.")Out [4]:
Using pad_token, but it is not set yet.
huggingface/tokenizers: The current process just got forked, after parallelism has already been used. Disabling parallelism to avoid deadlocks... To disable this warning, you can either: - Avoid using `tokenizers` before the fork if possible - Explicitly set the environment variable TOKENIZERS_PARALLELISM=(true | false)
0%| | 0/1 [00:00<?, ?it/s]
{'score': 190.3675537109375}In [6]:
# The perplexity is much higher since LangChain was introduced after 'gpt-2' was released and because it is never used in the following context.
evaluator.evaluate_strings(prediction="The rains in Spain fall mainly on LangChain.")Out [6]:
Using pad_token, but it is not set yet.
0%| | 0/1 [00:00<?, ?it/s]
{'score': 1982.0709228515625}In [ ]: