codekingpro/portable-devtools
114k
1"""This module contains the evaluator classes for evaluating runs."""2 3from __future__ import annotations4 5import asyncio6import inspect7import logging8import uuid9from abc import abstractmethod10from collections.abc import Awaitable, Sequence11from functools import wraps12from typing import (13 Any,14 Callable,15 Literal,16 Optional,17 Union,18 cast,19)20 21from pydantic import BaseModel, ConfigDict, Field, ValidationError, model_validator22from typing_extensions import TypedDict23 24from langsmith import run_helpers as rh25from langsmith import schemas26from langsmith.schemas import SCORE_TYPE, VALUE_TYPE, Example, Run27 28logger = logging.getLogger(__name__)29 30 31class Category(TypedDict):32 """A category for categorical feedback."""33 34 value: Optional[Union[float, int]]35 """The numeric score/ordinal corresponding to this category."""36 label: str37 """The label for this category."""38 39 40class FeedbackConfig(TypedDict, total=False):41 """Configuration to define a type of feedback.42 43 Applied on on the first creation of a `feedback_key`.44 """45 46 type: Literal["continuous", "categorical", "freeform"]47 """The type of feedback."""48 min: Optional[Union[float, int]]49 """The minimum permitted value (if continuous type)."""50 max: Optional[Union[float, int]]51 """The maximum value permitted value (if continuous type)."""52 categories: Optional[list[Union[Category, dict]]]53 54 55class EvaluationResult(BaseModel):56 """Evaluation result."""57 58 key: str59 """The aspect, metric name, or label for this evaluation."""60 score: SCORE_TYPE = None61 """The numeric score for this evaluation."""62 value: VALUE_TYPE = None63 """The value for this evaluation, if not numeric."""64 metadata: Optional[dict] = None65 """Arbitrary metadata attached to the evaluation."""66 comment: Optional[str] = None67 """An explanation regarding the evaluation."""68 correction: Optional[dict] = None69 """What the correct value should be, if applicable."""70 evaluator_info: dict = Field(default_factory=dict)71 """Additional information about the evaluator."""72 feedback_config: Optional[Union[FeedbackConfig, dict]] = None73 """The configuration used to generate this feedback."""74 source_run_id: Optional[Union[uuid.UUID, str]] = None75 """The ID of the trace of the evaluator itself."""76 target_run_id: Optional[Union[uuid.UUID, str]] = None77 """The ID of the trace this evaluation is applied to.78 79 If none provided, the evaluation feedback is applied to the80 root trace being."""81 extra: Optional[dict] = None82 """Metadata for the evaluator run."""83 84 model_config = ConfigDict(extra="forbid")85 86 @model_validator(mode="after")87 def check_value_non_numeric(self) -> EvaluationResult:88 """Warn when numeric values are passed via the `value` field."""89 if self.score is None and isinstance(self.value, (int, float)):90 logger.warning(91 "Numeric values should be provided in the 'score' field, not 'value'."92 f" Got: {self.value}"93 )94 return self95 96 97class EvaluationResults(TypedDict, total=False):98 """Batch evaluation results.99 100 This makes it easy for your evaluator to return multiple101 metrics at once.102 """103 104 results: list[EvaluationResult]105 """The evaluation results."""106 107 108class RunEvaluator:109 """Evaluator interface class."""110 111 @abstractmethod112 def evaluate_run(113 self,114 run: Run,115 example: Optional[Example] = None,116 evaluator_run_id: Optional[uuid.UUID] = None,117 ) -> Union[EvaluationResult, EvaluationResults]:118 """Evaluate an example."""119 120 async def aevaluate_run(121 self,122 run: Run,123 example: Optional[Example] = None,124 evaluator_run_id: Optional[uuid.UUID] = None,125 ) -> Union[EvaluationResult, EvaluationResults]:126 """Evaluate an example asynchronously."""127 current_context = rh.get_tracing_context()128 129 def _run_with_context():130 with rh.tracing_context(**current_context):131 return self.evaluate_run(run, example, evaluator_run_id)132 133 return await asyncio.get_running_loop().run_in_executor(None, _run_with_context)134 135 136_RUNNABLE_OUTPUT = Union[EvaluationResult, EvaluationResults, dict]137 138 139class ComparisonEvaluationResult(BaseModel):140 """Feedback scores for the results of comparative evaluations.141 142 These are generated by functions that compare two or more runs,143 returning a ranking or other feedback.144 """145 146 key: str147 """The aspect, metric name, or label for this evaluation."""148 scores: dict[Union[uuid.UUID, str], SCORE_TYPE]149 """The scores for each run in the comparison."""150 source_run_id: Optional[Union[uuid.UUID, str]] = None151 """The ID of the trace of the evaluator itself."""152 comment: Optional[Union[str, dict[Union[uuid.UUID, str], str]]] = None153 """Comment for the scores. If a string, it's shared across all target runs.154 155 If a `dict`, it maps run IDs to individual comments.156 """157 158 159_COMPARISON_OUTPUT = Union[ComparisonEvaluationResult, dict]160 161 162class DynamicRunEvaluator(RunEvaluator):163 """A dynamic evaluator that wraps a function and transforms it into a `RunEvaluator`.164 165 This class is designed to be used with the `@run_evaluator` decorator, allowing166 functions that take a `Run` and an optional `Example` as arguments, and return167 an `EvaluationResult` or `EvaluationResults`, to be used as instances of `RunEvaluator`.168 169 Attributes:170 func (Callable): The function that is wrapped by this evaluator.171 """ # noqa: E501172 173 def __init__(174 self,175 func: Callable[176 [Run, Optional[Example]],177 Union[_RUNNABLE_OUTPUT, Awaitable[_RUNNABLE_OUTPUT]],178 ],179 # Async function to be used for async evaluation. Optional180 afunc: Optional[181 Callable[182 [Run, Optional[Example]],183 Awaitable[_RUNNABLE_OUTPUT],184 ]185 ] = None,186 ):187 """Initialize the `DynamicRunEvaluator` with a given function.188 189 Args:190 func (Callable): A function that takes a `Run` and an optional `Example` as191 arguments, and returns a dict or `ComparisonEvaluationResult`.192 """193 (func, prepare_inputs) = _normalize_evaluator_func(func)194 if afunc:195 (afunc, prepare_inputs) = _normalize_evaluator_func(afunc) # type: ignore[assignment]196 197 def process_inputs(inputs: dict) -> dict:198 if prepare_inputs is None:199 return inputs200 (_, _, traced_inputs) = prepare_inputs(201 inputs.get("run"), inputs.get("example")202 )203 return traced_inputs204 205 wraps(func)(self)206 from langsmith import run_helpers # type: ignore207 208 if afunc is not None:209 self.afunc = run_helpers.ensure_traceable(210 afunc, process_inputs=process_inputs211 )212 self._name = getattr(afunc, "__name__", "DynamicRunEvaluator")213 if inspect.iscoroutinefunction(func):214 if afunc is not None:215 raise TypeError(216 "Func was provided as a coroutine function, but afunc was "217 "also provided. If providing both, func should be a regular "218 "function to avoid ambiguity."219 )220 self.afunc = run_helpers.ensure_traceable(221 func, process_inputs=process_inputs222 )223 self._name = getattr(func, "__name__", "DynamicRunEvaluator")224 else:225 self.func = run_helpers.ensure_traceable(226 cast(Callable[[Run, Optional[Example]], _RUNNABLE_OUTPUT], func),227 process_inputs=process_inputs,228 )229 self._name = getattr(func, "__name__", "DynamicRunEvaluator")230 231 def _coerce_evaluation_result(232 self,233 result: Union[EvaluationResult, dict],234 source_run_id: uuid.UUID,235 allow_no_key: bool = False,236 ) -> EvaluationResult:237 if isinstance(result, EvaluationResult):238 if not result.source_run_id:239 result.source_run_id = source_run_id240 return result241 try:242 if not result:243 raise ValueError(244 "Expected an EvaluationResult object, or dict with a metric"245 f" 'key' and optional 'score'; got empty result: {result}"246 )247 if "key" not in result and allow_no_key:248 result["key"] = self._name249 if all(k not in result for k in ("score", "value", "comment")):250 raise ValueError(251 "Expected an EvaluationResult object, or dict with a metric"252 f" 'key' and optional 'score' or categorical 'value'; got {result}"253 )254 return EvaluationResult(**{"source_run_id": source_run_id, **result})255 except ValidationError as e:256 raise ValueError(257 "Expected an EvaluationResult object, or dict with a metric"258 f" 'key' and optional 'score'; got {result}"259 ) from e260 261 def _coerce_evaluation_results(262 self,263 results: Union[dict, EvaluationResults],264 source_run_id: uuid.UUID,265 ) -> Union[EvaluationResult, EvaluationResults]:266 if "results" in results:267 cp = results.copy()268 cp["results"] = [269 self._coerce_evaluation_result(r, source_run_id=source_run_id)270 for r in results["results"]271 ]272 return EvaluationResults(**cp)273 274 return self._coerce_evaluation_result(275 cast(dict, results), source_run_id=source_run_id, allow_no_key=True276 )277 278 def _format_result(279 self,280 result: Union[281 EvaluationResult, EvaluationResults, dict, str, int, bool, float, list282 ],283 source_run_id: uuid.UUID,284 ) -> Union[EvaluationResult, EvaluationResults]:285 if isinstance(result, EvaluationResult):286 if not result.source_run_id:287 result.source_run_id = source_run_id288 return result289 result = _format_evaluator_result(result)290 return self._coerce_evaluation_results(result, source_run_id)291 292 @property293 def is_async(self) -> bool:294 """Check if the evaluator function is asynchronous.295 296 Returns:297 bool: `True` if the evaluator function is asynchronous, `False` otherwise.298 """299 return hasattr(self, "afunc")300 301 def evaluate_run(302 self,303 run: Run,304 example: Optional[Example] = None,305 evaluator_run_id: Optional[uuid.UUID] = None,306 ) -> Union[EvaluationResult, EvaluationResults]:307 """Evaluate a run using the wrapped function.308 309 This method directly invokes the wrapped function with the provided arguments.310 311 Args:312 run (Run): The run to be evaluated.313 example (Optional[Example]): An optional example to be used in the evaluation.314 315 Returns:316 Union[EvaluationResult, EvaluationResults]: The result of the evaluation.317 """ # noqa: E501318 if not hasattr(self, "func"):319 running_loop = asyncio.get_event_loop()320 if running_loop.is_running():321 raise RuntimeError(322 "Cannot call `evaluate_run` on an async run evaluator from"323 " within an running event loop. Use `aevaluate_run` instead."324 )325 else:326 return running_loop.run_until_complete(self.aevaluate_run(run, example))327 if evaluator_run_id is None:328 evaluator_run_id = uuid.uuid4()329 metadata: dict[str, Any] = {"target_run_id": run.id}330 if getattr(run, "session_id", None):331 metadata["experiment"] = str(run.session_id)332 result = self.func(333 run,334 example,335 langsmith_extra={"run_id": evaluator_run_id, "metadata": metadata},336 )337 return self._format_result(result, evaluator_run_id)338 339 async def aevaluate_run(340 self,341 run: Run,342 example: Optional[Example] = None,343 evaluator_run_id: Optional[uuid.UUID] = None,344 ):345 """Evaluate a run asynchronously using the wrapped async function.346 347 This method directly invokes the wrapped async function with the348 provided arguments.349 350 Args:351 run (Run): The run to be evaluated.352 example (Optional[Example]): An optional example to be used353 in the evaluation.354 355 Returns:356 Union[EvaluationResult, EvaluationResults]: The result of the evaluation.357 """358 if not hasattr(self, "afunc"):359 return await super().aevaluate_run(run, example)360 if evaluator_run_id is None:361 evaluator_run_id = uuid.uuid4()362 metadata: dict[str, Any] = {"target_run_id": run.id}363 if getattr(run, "session_id", None):364 metadata["experiment"] = str(run.session_id)365 result = await self.afunc(366 run,367 example,368 langsmith_extra={"run_id": evaluator_run_id, "metadata": metadata},369 )370 return self._format_result(result, evaluator_run_id)371 372 def __call__(373 self, run: Run, example: Optional[Example] = None374 ) -> Union[EvaluationResult, EvaluationResults]:375 """Make the evaluator callable, allowing it to be used like a function.376 377 This method enables the evaluator instance to be called directly, forwarding the378 call to `evaluate_run`.379 380 Args:381 run (Run): The run to be evaluated.382 example (Optional[Example]): An optional example to be used in the evaluation.383 384 Returns:385 Union[EvaluationResult, EvaluationResults]: The result of the evaluation.386 """ # noqa: E501387 return self.evaluate_run(run, example)388 389 def __repr__(self) -> str:390 """Represent the DynamicRunEvaluator object."""391 return f"<DynamicRunEvaluator {self._name}>"392 393 394def run_evaluator(395 func: Callable[396 [Run, Optional[Example]], Union[_RUNNABLE_OUTPUT, Awaitable[_RUNNABLE_OUTPUT]]397 ],398):399 """Create a run evaluator from a function.400 401 Decorator that transforms a function into a `RunEvaluator`.402 """403 return DynamicRunEvaluator(func)404 405 406_MAXSIZE = 10_000407 408 409def _maxsize_repr(obj: Any):410 s = repr(obj)411 if len(s) > _MAXSIZE:412 s = s[: _MAXSIZE - 4] + "...)"413 return s414 415 416class DynamicComparisonRunEvaluator:417 """Compare predictions (as traces) from 2 or more runs."""418 419 def __init__(420 self,421 func: Callable[422 [Sequence[Run], Optional[Example]],423 Union[_COMPARISON_OUTPUT, Awaitable[_COMPARISON_OUTPUT]],424 ],425 # Async function to be used for async evaluation. Optional426 afunc: Optional[427 Callable[428 [Sequence[Run], Optional[Example]],429 Awaitable[_COMPARISON_OUTPUT],430 ]431 ] = None,432 ):433 """Initialize the `DynamicRunEvaluator` with a given function.434 435 Args:436 func (Callable): A function that takes a `Run` and an optional `Example` as437 arguments, and returns an `EvaluationResult` or `EvaluationResults`.438 """439 (func, prepare_inputs) = _normalize_comparison_evaluator_func(func)440 if afunc:441 (afunc, prepare_inputs) = _normalize_comparison_evaluator_func(afunc) # type: ignore[assignment]442 443 def process_inputs(inputs: dict) -> dict:444 if prepare_inputs is None:445 return inputs446 (_, _, traced_inputs) = prepare_inputs(447 inputs.get("runs"), inputs.get("example")448 )449 return traced_inputs450 451 wraps(func)(self)452 from langsmith import run_helpers # type: ignore453 454 if afunc is not None:455 self.afunc = run_helpers.ensure_traceable(456 afunc, process_inputs=process_inputs457 )458 self._name = getattr(afunc, "__name__", "DynamicRunEvaluator")459 if inspect.iscoroutinefunction(func):460 if afunc is not None:461 raise TypeError(462 "Func was provided as a coroutine function, but afunc was "463 "also provided. If providing both, func should be a regular "464 "function to avoid ambiguity."465 )466 self.afunc = run_helpers.ensure_traceable(467 func, process_inputs=process_inputs468 )469 self._name = getattr(func, "__name__", "DynamicRunEvaluator")470 else:471 self.func = run_helpers.ensure_traceable(472 cast(473 Callable[474 [Sequence[Run], Optional[Example]],475 _COMPARISON_OUTPUT,476 ],477 func,478 ),479 process_inputs=process_inputs,480 )481 self._name = getattr(func, "__name__", "DynamicRunEvaluator")482 483 @property484 def is_async(self) -> bool:485 """Check if the evaluator function is asynchronous.486 487 Returns:488 bool: `True` if the evaluator function is asynchronous, `False` otherwise.489 """490 return hasattr(self, "afunc")491 492 def compare_runs(493 self, runs: Sequence[Run], example: Optional[Example] = None494 ) -> ComparisonEvaluationResult:495 """Compare runs to score preferences.496 497 Args:498 runs: A list of runs to compare.499 example: An optional example to be used in the evaluation.500 501 """ # noqa: E501502 if not hasattr(self, "func"):503 running_loop = asyncio.get_event_loop()504 if running_loop.is_running():505 raise RuntimeError(506 "Cannot call `evaluate_run` on an async run evaluator from"507 " within an running event loop. Use `aevaluate_run` instead."508 )509 else:510 return running_loop.run_until_complete(511 self.acompare_runs(runs, example)512 )513 source_run_id = uuid.uuid4()514 tags = self._get_tags(runs)515 # TODO: Add metadata for the "comparison experiment" here516 result = self.func(517 runs,518 example,519 langsmith_extra={"run_id": source_run_id, "tags": tags},520 )521 return self._format_results(result, source_run_id, runs)522 523 async def acompare_runs(524 self, runs: Sequence[Run], example: Optional[Example] = None525 ) -> ComparisonEvaluationResult:526 """Evaluate a run asynchronously using the wrapped async function.527 528 This method directly invokes the wrapped async function with the529 provided arguments.530 531 Args:532 runs (Run): The runs to be evaluated.533 example (Optional[Example]): An optional example to be used534 in the evaluation.535 536 Returns:537 ComparisonEvaluationResult: The result of the evaluation.538 """539 if not hasattr(self, "afunc"):540 return self.compare_runs(runs, example)541 source_run_id = uuid.uuid4()542 tags = self._get_tags(runs)543 # TODO: Add metadata for the "comparison experiment" here544 result = await self.afunc(545 runs,546 example,547 langsmith_extra={"run_id": source_run_id, "tags": tags},548 )549 return self._format_results(result, source_run_id, runs)550 551 def __call__(552 self, runs: Sequence[Run], example: Optional[Example] = None553 ) -> ComparisonEvaluationResult:554 """Make the evaluator callable, allowing it to be used like a function.555 556 This method enables the evaluator instance to be called directly, forwarding the557 call to `evaluate_run`.558 559 Args:560 run (Run): The run to be evaluated.561 example (Optional[Example]): An optional example to be used in the evaluation.562 563 Returns:564 ComparisonEvaluationResult: The result of the evaluation.565 """ # noqa: E501566 return self.compare_runs(runs, example)567 568 def __repr__(self) -> str:569 """Represent the DynamicRunEvaluator object."""570 return f"<DynamicComparisonRunEvaluator {self._name}>"571 572 @staticmethod573 def _get_tags(runs: Sequence[Run]) -> list[str]:574 """Extract tags from runs."""575 # Add tags to support filtering576 tags = []577 for run in runs:578 tags.append("run:" + str(run.id))579 if getattr(run, "session_id", None):580 tags.append("experiment:" + str(run.session_id))581 return tags582 583 def _format_results(584 self,585 result: Union[dict, list, ComparisonEvaluationResult],586 source_run_id: uuid.UUID,587 runs: Sequence[Run],588 ) -> ComparisonEvaluationResult:589 if isinstance(result, ComparisonEvaluationResult):590 if not result.source_run_id:591 result.source_run_id = source_run_id592 return result593 elif isinstance(result, list):594 result = {595 "scores": {run.id: score for run, score in zip(runs, result)},596 "key": self._name,597 "source_run_id": source_run_id,598 }599 elif isinstance(result, dict):600 if "key" not in result:601 result["key"] = self._name602 else:603 msg = (604 "Expected 'dict', 'list' or 'ComparisonEvaluationResult' result "605 f"object. Received: {result=}"606 )607 raise ValueError(msg)608 try:609 return ComparisonEvaluationResult(610 **{"source_run_id": source_run_id, **result}611 )612 except ValidationError as e:613 raise ValueError(614 f"Expected a dictionary with a 'key' and dictionary of scores mapping"615 "run IDs to numeric scores, or ComparisonEvaluationResult object,"616 f" got {result}"617 ) from e618 619 620def comparison_evaluator(621 func: Callable[622 [Sequence[Run], Optional[Example]],623 Union[_COMPARISON_OUTPUT, Awaitable[_COMPARISON_OUTPUT]],624 ],625) -> DynamicComparisonRunEvaluator:626 """Create a comaprison evaluator from a function."""627 return DynamicComparisonRunEvaluator(func)628 629 630def _normalize_evaluator_func(631 func: Callable,632) -> tuple[633 Union[634 Callable[[Run, Optional[Example]], _RUNNABLE_OUTPUT],635 Callable[[Run, Optional[Example]], Awaitable[_RUNNABLE_OUTPUT]],636 ],637 Optional[Callable[..., dict]],638]:639 supported_args = (640 "run",641 "example",642 "inputs",643 "outputs",644 "reference_outputs",645 "attachments",646 )647 sig = inspect.signature(func)648 all_args = [pname for pname, p in sig.parameters.items() if p.kind != p.VAR_KEYWORD]649 args_with_defaults = [650 pname651 for pname, p in sig.parameters.items()652 if p.default is not inspect.Parameter.empty653 ]654 if not all_args or (655 not all(656 pname in supported_args or pname in args_with_defaults for pname in all_args657 )658 and len([a for a in all_args if a not in args_with_defaults]) != 2659 ):660 msg = (661 f"Invalid evaluator function. Must have at least one "662 f"argument. Supported arguments are {supported_args}. Please "663 f"see https://docs.smith.langchain.com/evaluation/how_to_guides/evaluation/evaluate_llm_application#use-custom-evaluators"664 # noqa: E501665 )666 raise ValueError(msg)667 # For backwards compatibility we assume custom arg names are Run and Example668 # types, respectively.669 elif not all(670 pname in supported_args or pname in args_with_defaults for pname in all_args671 ) or all_args == [672 "run",673 "example",674 ]:675 return func, None676 else:677 if inspect.iscoroutinefunction(func):678 679 def _prepare_inputs(680 run: Run, example: Optional[Example]681 ) -> tuple[list, dict, dict]:682 arg_map = {683 "run": run,684 "example": example,685 "inputs": example.inputs if example else {},686 "outputs": run.outputs or {},687 "attachments": example.attachments or {} if example else {},688 "reference_outputs": example.outputs or {} if example else {},689 }690 kwargs = {}691 args = []692 traced_inputs = {}693 for param_name, param in sig.parameters.items():694 # Could have params with defaults that are not in the arg map695 if param_name in arg_map:696 if param.kind in (697 param.POSITIONAL_OR_KEYWORD,698 param.POSITIONAL_ONLY,699 ):700 args.append(arg_map[param_name])701 else:702 kwargs[param_name] = arg_map[param_name]703 traced_inputs[param_name] = (704 _maxsize_repr(arg_map[param_name])705 if param_name in ("run", "example")706 else arg_map[param_name]707 )708 return args, kwargs, traced_inputs709 710 async def awrapper(711 run: Run, example: Optional[Example]712 ) -> _RUNNABLE_OUTPUT:713 (args, kwargs, _) = _prepare_inputs(run, example)714 return await func(*args, **kwargs)715 716 awrapper.__name__ = (717 getattr(func, "__name__")718 if hasattr(func, "__name__")719 else awrapper.__name__720 )721 return (awrapper, _prepare_inputs) # type: ignore[return-value]722 723 else:724 725 def _prepare_inputs(726 run: Run, example: Optional[Example]727 ) -> tuple[list, dict, dict]:728 arg_map = {729 "run": run,730 "example": example,731 "inputs": example.inputs if example else {},732 "outputs": run.outputs or {},733 "attachments": example.attachments or {} if example else {},734 "reference_outputs": example.outputs or {} if example else {},735 }736 kwargs = {}737 args = []738 traced_inputs = {}739 for param_name, param in sig.parameters.items():740 # Could have params with defaults that are not in the arg map741 if param_name in arg_map:742 if param.kind in (743 param.POSITIONAL_OR_KEYWORD,744 param.POSITIONAL_ONLY,745 ):746 args.append(arg_map[param_name])747 else:748 kwargs[param_name] = arg_map[param_name]749 traced_inputs[param_name] = (750 _maxsize_repr(arg_map[param_name])751 if param_name in ("run", "example")752 else arg_map[param_name]753 )754 return args, kwargs, traced_inputs755 756 def wrapper(run: Run, example: Optional[Example]) -> _RUNNABLE_OUTPUT:757 (args, kwargs, _) = _prepare_inputs(run, example)758 return func(*args, **kwargs)759 760 wrapper.__name__ = (761 getattr(func, "__name__")762 if hasattr(func, "__name__")763 else wrapper.__name__764 )765 return (wrapper, _prepare_inputs) # type: ignore[return-value]766 767 768def _normalize_comparison_evaluator_func(769 func: Callable,770) -> tuple[771 Union[772 Callable[[Sequence[Run], Optional[Example]], _COMPARISON_OUTPUT],773 Callable[[Sequence[Run], Optional[Example]], Awaitable[_COMPARISON_OUTPUT]],774 ],775 Optional[Callable[..., dict]],776]:777 supported_args = ("runs", "example", "inputs", "outputs", "reference_outputs")778 sig = inspect.signature(func)779 all_args = [pname for pname, p in sig.parameters.items() if p.kind != p.VAR_KEYWORD]780 args_with_defaults = [781 pname782 for pname, p in sig.parameters.items()783 if p.default is not inspect.Parameter.empty784 ]785 if not all_args or (786 not all(787 pname in supported_args or pname in args_with_defaults for pname in all_args788 )789 and len([a for a in all_args if a not in args_with_defaults]) != 2790 ):791 msg = (792 f"Invalid evaluator function. Must have at least one "793 f"argument. Supported arguments are {supported_args}. Please "794 f"see https://docs.smith.langchain.com/evaluation/how_to_guides/evaluation/evaluate_llm_application#use-custom-evaluators"795 # noqa: E501796 )797 raise ValueError(msg)798 # For backwards compatibility we assume custom arg names are List[Run] and799 # List[Example] types, respectively.800 elif not all(801 pname in supported_args or pname in args_with_defaults for pname in all_args802 ) or all_args == [803 "runs",804 "example",805 ]:806 return func, None807 else:808 if inspect.iscoroutinefunction(func):809 810 def _prepare_inputs(811 runs: Sequence[Run], example: Optional[Example]812 ) -> tuple[list, dict, dict]:813 arg_map = {814 "runs": runs,815 "example": example,816 "inputs": example.inputs if example else {},817 "outputs": [run.outputs or {} for run in runs],818 "reference_outputs": example.outputs or {} if example else {},819 }820 kwargs = {}821 args = []822 traced_inputs = {}823 for param_name, param in sig.parameters.items():824 # Could have params with defaults that are not in the arg map825 if param_name in arg_map:826 if param.kind in (827 param.POSITIONAL_OR_KEYWORD,828 param.POSITIONAL_ONLY,829 ):830 args.append(arg_map[param_name])831 else:832 kwargs[param_name] = arg_map[param_name]833 traced_inputs[param_name] = (834 _maxsize_repr(arg_map[param_name])835 if param_name in ("runs", "example")836 else arg_map[param_name]837 )838 return args, kwargs, traced_inputs839 840 async def awrapper(841 runs: Sequence[Run], example: Optional[Example]842 ) -> _COMPARISON_OUTPUT:843 (args, kwargs, _) = _prepare_inputs(runs, example)844 return await func(*args, **kwargs)845 846 awrapper.__name__ = (847 getattr(func, "__name__")848 if hasattr(func, "__name__")849 else awrapper.__name__850 )851 return awrapper, _prepare_inputs # type: ignore[return-value]852 853 else:854 855 def _prepare_inputs(856 runs: Sequence[Run], example: Optional[Example]857 ) -> tuple[list, dict, dict]:858 arg_map = {859 "runs": runs,860 "example": example,861 "inputs": example.inputs if example else {},862 "outputs": [run.outputs or {} for run in runs],863 "reference_outputs": example.outputs or {} if example else {},864 }865 kwargs = {}866 args = []867 traced_inputs = {}868 for param_name, param in sig.parameters.items():869 # Could have params with defaults that are not in the arg map870 if param_name in arg_map:871 if param.kind in (872 param.POSITIONAL_OR_KEYWORD,873 param.POSITIONAL_ONLY,874 ):875 args.append(arg_map[param_name])876 else:877 kwargs[param_name] = arg_map[param_name]878 traced_inputs[param_name] = (879 _maxsize_repr(arg_map[param_name])880 if param_name in ("runs", "example")881 else arg_map[param_name]882 )883 return args, kwargs, traced_inputs884 885 def wrapper(886 runs: Sequence[Run], example: Optional[Example]887 ) -> _COMPARISON_OUTPUT:888 (args, kwargs, _) = _prepare_inputs(runs, example)889 return func(*args, **kwargs)890 891 wrapper.__name__ = (892 getattr(func, "__name__")893 if hasattr(func, "__name__")894 else wrapper.__name__895 )896 return wrapper, _prepare_inputs # type: ignore[return-value]897 898 899def _format_evaluator_result(900 result: Union[EvaluationResults, dict, str, int, bool, float, list],901) -> Union[EvaluationResults, dict]:902 if isinstance(result, (bool, float, int)):903 result = {"score": result}904 elif not result:905 raise ValueError(906 f"Expected a non-empty dict, str, bool, int, float, list, "907 f"EvaluationResult, or EvaluationResults. Got {result}"908 )909 elif isinstance(result, list):910 if not all(isinstance(x, dict) for x in result):911 raise ValueError(912 f"Expected a list of dicts or EvaluationResults. Received {result}."913 )914 result = {"results": result} # type: ignore[misc]915 elif isinstance(result, str):916 result = {"value": result}917 elif isinstance(result, dict):918 pass919 else:920 raise ValueError(921 f"Expected a dict, str, bool, int, float, list, EvaluationResult, or "922 f"EvaluationResults. Got {result}"923 )924 return result925 926 927SUMMARY_EVALUATOR_T = Union[928 Callable[929 [Sequence[schemas.Run], Sequence[schemas.Example]],930 Union[EvaluationResult, EvaluationResults],931 ],932 Callable[933 [list[schemas.Run], list[schemas.Example]],934 Union[EvaluationResult, EvaluationResults],935 ],936]937 938 939def _normalize_summary_evaluator(func: Callable) -> SUMMARY_EVALUATOR_T:940 supported_args = ("runs", "examples", "inputs", "outputs", "reference_outputs")941 sig = inspect.signature(func)942 all_args = [pname for pname, p in sig.parameters.items()]943 args_with_defaults = [944 pname945 for pname, p in sig.parameters.items()946 if p.default is not inspect.Parameter.empty947 ]948 if not all_args or (949 not all(950 pname in supported_args or pname in args_with_defaults for pname in all_args951 )952 and len([a for a in all_args if a not in args_with_defaults]) != 2953 ):954 msg = (955 f"Invalid evaluator function. Must have at least one "956 f"argument. Supported arguments are {supported_args}."957 )958 if all_args:959 msg += f" Received arguments {all_args}."960 raise ValueError(msg)961 # For backwards compatibility we assume custom arg names are Sequence[Run] and962 # Sequence[Example] types, respectively.963 elif not all(pname in supported_args for pname in all_args) or all_args == [964 "runs",965 "examples",966 ]:967 return func968 else:969 970 def wrapper(971 runs: Sequence[schemas.Run], examples: Sequence[schemas.Example]972 ) -> Union[EvaluationResult, EvaluationResults]:973 arg_map = {974 "runs": runs,975 "examples": examples,976 "inputs": [example.inputs for example in examples],977 "outputs": [run.outputs or {} for run in runs],978 "reference_outputs": [example.outputs or {} for example in examples],979 }980 kwargs = {}981 args = []982 for param_name, param in sig.parameters.items():983 # Could have params with defaults that are not in the arg map984 if param_name in arg_map:985 if param.kind in (986 param.POSITIONAL_OR_KEYWORD,987 param.POSITIONAL_ONLY,988 ):989 args.append(arg_map[param_name])990 else:991 kwargs[param_name] = arg_map[param_name]992 993 result = func(*args, **kwargs)994 if isinstance(result, EvaluationResult):995 return result996 return _format_evaluator_result(result) # type: ignore997 998 wrapper.__name__ = (999 getattr(func, "__name__") if hasattr(func, "__name__") else wrapper.__name__1000 )1001 return wrapper # type: ignore[return-value]1002 