Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
evaluator.py1002 linesDownload Raw Back to evaluation
1"""This module contains the evaluator classes for evaluating runs."""2 3from __future__ import annotations4 5import asyncio6import inspect7import logging8import uuid9from abc import abstractmethod10from collections.abc import Awaitable, Sequence11from functools import wraps12from typing import (13    Any,14    Callable,15    Literal,16    Optional,17    Union,18    cast,19)20 21from pydantic import BaseModel, ConfigDict, Field, ValidationError, model_validator22from typing_extensions import TypedDict23 24from langsmith import run_helpers as rh25from langsmith import schemas26from langsmith.schemas import SCORE_TYPE, VALUE_TYPE, Example, Run27 28logger = logging.getLogger(__name__)29 30 31class Category(TypedDict):32    """A category for categorical feedback."""33 34    value: Optional[Union[float, int]]35    """The numeric score/ordinal corresponding to this category."""36    label: str37    """The label for this category."""38 39 40class FeedbackConfig(TypedDict, total=False):41    """Configuration to define a type of feedback.42 43    Applied on on the first creation of a `feedback_key`.44    """45 46    type: Literal["continuous", "categorical", "freeform"]47    """The type of feedback."""48    min: Optional[Union[float, int]]49    """The minimum permitted value (if continuous type)."""50    max: Optional[Union[float, int]]51    """The maximum value permitted value (if continuous type)."""52    categories: Optional[list[Union[Category, dict]]]53 54 55class EvaluationResult(BaseModel):56    """Evaluation result."""57 58    key: str59    """The aspect, metric name, or label for this evaluation."""60    score: SCORE_TYPE = None61    """The numeric score for this evaluation."""62    value: VALUE_TYPE = None63    """The value for this evaluation, if not numeric."""64    metadata: Optional[dict] = None65    """Arbitrary metadata attached to the evaluation."""66    comment: Optional[str] = None67    """An explanation regarding the evaluation."""68    correction: Optional[dict] = None69    """What the correct value should be, if applicable."""70    evaluator_info: dict = Field(default_factory=dict)71    """Additional information about the evaluator."""72    feedback_config: Optional[Union[FeedbackConfig, dict]] = None73    """The configuration used to generate this feedback."""74    source_run_id: Optional[Union[uuid.UUID, str]] = None75    """The ID of the trace of the evaluator itself."""76    target_run_id: Optional[Union[uuid.UUID, str]] = None77    """The ID of the trace this evaluation is applied to.78 79    If none provided, the evaluation feedback is applied to the80    root trace being."""81    extra: Optional[dict] = None82    """Metadata for the evaluator run."""83 84    model_config = ConfigDict(extra="forbid")85 86    @model_validator(mode="after")87    def check_value_non_numeric(self) -> EvaluationResult:88        """Warn when numeric values are passed via the `value` field."""89        if self.score is None and isinstance(self.value, (int, float)):90            logger.warning(91                "Numeric values should be provided in the 'score' field, not 'value'."92                f" Got: {self.value}"93            )94        return self95 96 97class EvaluationResults(TypedDict, total=False):98    """Batch evaluation results.99 100    This makes it easy for your evaluator to return multiple101    metrics at once.102    """103 104    results: list[EvaluationResult]105    """The evaluation results."""106 107 108class RunEvaluator:109    """Evaluator interface class."""110 111    @abstractmethod112    def evaluate_run(113        self,114        run: Run,115        example: Optional[Example] = None,116        evaluator_run_id: Optional[uuid.UUID] = None,117    ) -> Union[EvaluationResult, EvaluationResults]:118        """Evaluate an example."""119 120    async def aevaluate_run(121        self,122        run: Run,123        example: Optional[Example] = None,124        evaluator_run_id: Optional[uuid.UUID] = None,125    ) -> Union[EvaluationResult, EvaluationResults]:126        """Evaluate an example asynchronously."""127        current_context = rh.get_tracing_context()128 129        def _run_with_context():130            with rh.tracing_context(**current_context):131                return self.evaluate_run(run, example, evaluator_run_id)132 133        return await asyncio.get_running_loop().run_in_executor(None, _run_with_context)134 135 136_RUNNABLE_OUTPUT = Union[EvaluationResult, EvaluationResults, dict]137 138 139class ComparisonEvaluationResult(BaseModel):140    """Feedback scores for the results of comparative evaluations.141 142    These are generated by functions that compare two or more runs,143    returning a ranking or other feedback.144    """145 146    key: str147    """The aspect, metric name, or label for this evaluation."""148    scores: dict[Union[uuid.UUID, str], SCORE_TYPE]149    """The scores for each run in the comparison."""150    source_run_id: Optional[Union[uuid.UUID, str]] = None151    """The ID of the trace of the evaluator itself."""152    comment: Optional[Union[str, dict[Union[uuid.UUID, str], str]]] = None153    """Comment for the scores. If a string, it's shared across all target runs.154    155    If a `dict`, it maps run IDs to individual comments.156    """157 158 159_COMPARISON_OUTPUT = Union[ComparisonEvaluationResult, dict]160 161 162class DynamicRunEvaluator(RunEvaluator):163    """A dynamic evaluator that wraps a function and transforms it into a `RunEvaluator`.164 165    This class is designed to be used with the `@run_evaluator` decorator, allowing166    functions that take a `Run` and an optional `Example` as arguments, and return167    an `EvaluationResult` or `EvaluationResults`, to be used as instances of `RunEvaluator`.168 169    Attributes:170        func (Callable): The function that is wrapped by this evaluator.171    """  # noqa: E501172 173    def __init__(174        self,175        func: Callable[176            [Run, Optional[Example]],177            Union[_RUNNABLE_OUTPUT, Awaitable[_RUNNABLE_OUTPUT]],178        ],179        # Async function to be used for async evaluation. Optional180        afunc: Optional[181            Callable[182                [Run, Optional[Example]],183                Awaitable[_RUNNABLE_OUTPUT],184            ]185        ] = None,186    ):187        """Initialize the `DynamicRunEvaluator` with a given function.188 189        Args:190            func (Callable): A function that takes a `Run` and an optional `Example` as191            arguments, and returns a dict or `ComparisonEvaluationResult`.192        """193        (func, prepare_inputs) = _normalize_evaluator_func(func)194        if afunc:195            (afunc, prepare_inputs) = _normalize_evaluator_func(afunc)  # type: ignore[assignment]196 197        def process_inputs(inputs: dict) -> dict:198            if prepare_inputs is None:199                return inputs200            (_, _, traced_inputs) = prepare_inputs(201                inputs.get("run"), inputs.get("example")202            )203            return traced_inputs204 205        wraps(func)(self)206        from langsmith import run_helpers  # type: ignore207 208        if afunc is not None:209            self.afunc = run_helpers.ensure_traceable(210                afunc, process_inputs=process_inputs211            )212            self._name = getattr(afunc, "__name__", "DynamicRunEvaluator")213        if inspect.iscoroutinefunction(func):214            if afunc is not None:215                raise TypeError(216                    "Func was provided as a coroutine function, but afunc was "217                    "also provided. If providing both, func should be a regular "218                    "function to avoid ambiguity."219                )220            self.afunc = run_helpers.ensure_traceable(221                func, process_inputs=process_inputs222            )223            self._name = getattr(func, "__name__", "DynamicRunEvaluator")224        else:225            self.func = run_helpers.ensure_traceable(226                cast(Callable[[Run, Optional[Example]], _RUNNABLE_OUTPUT], func),227                process_inputs=process_inputs,228            )229            self._name = getattr(func, "__name__", "DynamicRunEvaluator")230 231    def _coerce_evaluation_result(232        self,233        result: Union[EvaluationResult, dict],234        source_run_id: uuid.UUID,235        allow_no_key: bool = False,236    ) -> EvaluationResult:237        if isinstance(result, EvaluationResult):238            if not result.source_run_id:239                result.source_run_id = source_run_id240            return result241        try:242            if not result:243                raise ValueError(244                    "Expected an EvaluationResult object, or dict with a metric"245                    f" 'key' and optional 'score'; got empty result: {result}"246                )247            if "key" not in result and allow_no_key:248                result["key"] = self._name249            if all(k not in result for k in ("score", "value", "comment")):250                raise ValueError(251                    "Expected an EvaluationResult object, or dict with a metric"252                    f" 'key' and optional 'score' or categorical 'value'; got {result}"253                )254            return EvaluationResult(**{"source_run_id": source_run_id, **result})255        except ValidationError as e:256            raise ValueError(257                "Expected an EvaluationResult object, or dict with a metric"258                f" 'key' and optional 'score'; got {result}"259            ) from e260 261    def _coerce_evaluation_results(262        self,263        results: Union[dict, EvaluationResults],264        source_run_id: uuid.UUID,265    ) -> Union[EvaluationResult, EvaluationResults]:266        if "results" in results:267            cp = results.copy()268            cp["results"] = [269                self._coerce_evaluation_result(r, source_run_id=source_run_id)270                for r in results["results"]271            ]272            return EvaluationResults(**cp)273 274        return self._coerce_evaluation_result(275            cast(dict, results), source_run_id=source_run_id, allow_no_key=True276        )277 278    def _format_result(279        self,280        result: Union[281            EvaluationResult, EvaluationResults, dict, str, int, bool, float, list282        ],283        source_run_id: uuid.UUID,284    ) -> Union[EvaluationResult, EvaluationResults]:285        if isinstance(result, EvaluationResult):286            if not result.source_run_id:287                result.source_run_id = source_run_id288            return result289        result = _format_evaluator_result(result)290        return self._coerce_evaluation_results(result, source_run_id)291 292    @property293    def is_async(self) -> bool:294        """Check if the evaluator function is asynchronous.295 296        Returns:297            bool: `True` if the evaluator function is asynchronous, `False` otherwise.298        """299        return hasattr(self, "afunc")300 301    def evaluate_run(302        self,303        run: Run,304        example: Optional[Example] = None,305        evaluator_run_id: Optional[uuid.UUID] = None,306    ) -> Union[EvaluationResult, EvaluationResults]:307        """Evaluate a run using the wrapped function.308 309        This method directly invokes the wrapped function with the provided arguments.310 311        Args:312            run (Run): The run to be evaluated.313            example (Optional[Example]): An optional example to be used in the evaluation.314 315        Returns:316            Union[EvaluationResult, EvaluationResults]: The result of the evaluation.317        """  # noqa: E501318        if not hasattr(self, "func"):319            running_loop = asyncio.get_event_loop()320            if running_loop.is_running():321                raise RuntimeError(322                    "Cannot call `evaluate_run` on an async run evaluator from"323                    " within an running event loop. Use `aevaluate_run` instead."324                )325            else:326                return running_loop.run_until_complete(self.aevaluate_run(run, example))327        if evaluator_run_id is None:328            evaluator_run_id = uuid.uuid4()329        metadata: dict[str, Any] = {"target_run_id": run.id}330        if getattr(run, "session_id", None):331            metadata["experiment"] = str(run.session_id)332        result = self.func(333            run,334            example,335            langsmith_extra={"run_id": evaluator_run_id, "metadata": metadata},336        )337        return self._format_result(result, evaluator_run_id)338 339    async def aevaluate_run(340        self,341        run: Run,342        example: Optional[Example] = None,343        evaluator_run_id: Optional[uuid.UUID] = None,344    ):345        """Evaluate a run asynchronously using the wrapped async function.346 347        This method directly invokes the wrapped async function with the348            provided arguments.349 350        Args:351            run (Run): The run to be evaluated.352            example (Optional[Example]): An optional example to be used353                in the evaluation.354 355        Returns:356            Union[EvaluationResult, EvaluationResults]: The result of the evaluation.357        """358        if not hasattr(self, "afunc"):359            return await super().aevaluate_run(run, example)360        if evaluator_run_id is None:361            evaluator_run_id = uuid.uuid4()362        metadata: dict[str, Any] = {"target_run_id": run.id}363        if getattr(run, "session_id", None):364            metadata["experiment"] = str(run.session_id)365        result = await self.afunc(366            run,367            example,368            langsmith_extra={"run_id": evaluator_run_id, "metadata": metadata},369        )370        return self._format_result(result, evaluator_run_id)371 372    def __call__(373        self, run: Run, example: Optional[Example] = None374    ) -> Union[EvaluationResult, EvaluationResults]:375        """Make the evaluator callable, allowing it to be used like a function.376 377        This method enables the evaluator instance to be called directly, forwarding the378        call to `evaluate_run`.379 380        Args:381            run (Run): The run to be evaluated.382            example (Optional[Example]): An optional example to be used in the evaluation.383 384        Returns:385            Union[EvaluationResult, EvaluationResults]: The result of the evaluation.386        """  # noqa: E501387        return self.evaluate_run(run, example)388 389    def __repr__(self) -> str:390        """Represent the DynamicRunEvaluator object."""391        return f"<DynamicRunEvaluator {self._name}>"392 393 394def run_evaluator(395    func: Callable[396        [Run, Optional[Example]], Union[_RUNNABLE_OUTPUT, Awaitable[_RUNNABLE_OUTPUT]]397    ],398):399    """Create a run evaluator from a function.400 401    Decorator that transforms a function into a `RunEvaluator`.402    """403    return DynamicRunEvaluator(func)404 405 406_MAXSIZE = 10_000407 408 409def _maxsize_repr(obj: Any):410    s = repr(obj)411    if len(s) > _MAXSIZE:412        s = s[: _MAXSIZE - 4] + "...)"413    return s414 415 416class DynamicComparisonRunEvaluator:417    """Compare predictions (as traces) from 2 or more runs."""418 419    def __init__(420        self,421        func: Callable[422            [Sequence[Run], Optional[Example]],423            Union[_COMPARISON_OUTPUT, Awaitable[_COMPARISON_OUTPUT]],424        ],425        # Async function to be used for async evaluation. Optional426        afunc: Optional[427            Callable[428                [Sequence[Run], Optional[Example]],429                Awaitable[_COMPARISON_OUTPUT],430            ]431        ] = None,432    ):433        """Initialize the `DynamicRunEvaluator` with a given function.434 435        Args:436            func (Callable): A function that takes a `Run` and an optional `Example` as437            arguments, and returns an `EvaluationResult` or `EvaluationResults`.438        """439        (func, prepare_inputs) = _normalize_comparison_evaluator_func(func)440        if afunc:441            (afunc, prepare_inputs) = _normalize_comparison_evaluator_func(afunc)  # type: ignore[assignment]442 443        def process_inputs(inputs: dict) -> dict:444            if prepare_inputs is None:445                return inputs446            (_, _, traced_inputs) = prepare_inputs(447                inputs.get("runs"), inputs.get("example")448            )449            return traced_inputs450 451        wraps(func)(self)452        from langsmith import run_helpers  # type: ignore453 454        if afunc is not None:455            self.afunc = run_helpers.ensure_traceable(456                afunc, process_inputs=process_inputs457            )458            self._name = getattr(afunc, "__name__", "DynamicRunEvaluator")459        if inspect.iscoroutinefunction(func):460            if afunc is not None:461                raise TypeError(462                    "Func was provided as a coroutine function, but afunc was "463                    "also provided. If providing both, func should be a regular "464                    "function to avoid ambiguity."465                )466            self.afunc = run_helpers.ensure_traceable(467                func, process_inputs=process_inputs468            )469            self._name = getattr(func, "__name__", "DynamicRunEvaluator")470        else:471            self.func = run_helpers.ensure_traceable(472                cast(473                    Callable[474                        [Sequence[Run], Optional[Example]],475                        _COMPARISON_OUTPUT,476                    ],477                    func,478                ),479                process_inputs=process_inputs,480            )481            self._name = getattr(func, "__name__", "DynamicRunEvaluator")482 483    @property484    def is_async(self) -> bool:485        """Check if the evaluator function is asynchronous.486 487        Returns:488            bool: `True` if the evaluator function is asynchronous, `False` otherwise.489        """490        return hasattr(self, "afunc")491 492    def compare_runs(493        self, runs: Sequence[Run], example: Optional[Example] = None494    ) -> ComparisonEvaluationResult:495        """Compare runs to score preferences.496 497        Args:498            runs: A list of runs to compare.499            example: An optional example to be used in the evaluation.500 501        """  # noqa: E501502        if not hasattr(self, "func"):503            running_loop = asyncio.get_event_loop()504            if running_loop.is_running():505                raise RuntimeError(506                    "Cannot call `evaluate_run` on an async run evaluator from"507                    " within an running event loop. Use `aevaluate_run` instead."508                )509            else:510                return running_loop.run_until_complete(511                    self.acompare_runs(runs, example)512                )513        source_run_id = uuid.uuid4()514        tags = self._get_tags(runs)515        # TODO: Add metadata for the "comparison experiment" here516        result = self.func(517            runs,518            example,519            langsmith_extra={"run_id": source_run_id, "tags": tags},520        )521        return self._format_results(result, source_run_id, runs)522 523    async def acompare_runs(524        self, runs: Sequence[Run], example: Optional[Example] = None525    ) -> ComparisonEvaluationResult:526        """Evaluate a run asynchronously using the wrapped async function.527 528        This method directly invokes the wrapped async function with the529            provided arguments.530 531        Args:532            runs (Run): The runs to be evaluated.533            example (Optional[Example]): An optional example to be used534                in the evaluation.535 536        Returns:537            ComparisonEvaluationResult: The result of the evaluation.538        """539        if not hasattr(self, "afunc"):540            return self.compare_runs(runs, example)541        source_run_id = uuid.uuid4()542        tags = self._get_tags(runs)543        # TODO: Add metadata for the "comparison experiment" here544        result = await self.afunc(545            runs,546            example,547            langsmith_extra={"run_id": source_run_id, "tags": tags},548        )549        return self._format_results(result, source_run_id, runs)550 551    def __call__(552        self, runs: Sequence[Run], example: Optional[Example] = None553    ) -> ComparisonEvaluationResult:554        """Make the evaluator callable, allowing it to be used like a function.555 556        This method enables the evaluator instance to be called directly, forwarding the557        call to `evaluate_run`.558 559        Args:560            run (Run): The run to be evaluated.561            example (Optional[Example]): An optional example to be used in the evaluation.562 563        Returns:564            ComparisonEvaluationResult: The result of the evaluation.565        """  # noqa: E501566        return self.compare_runs(runs, example)567 568    def __repr__(self) -> str:569        """Represent the DynamicRunEvaluator object."""570        return f"<DynamicComparisonRunEvaluator {self._name}>"571 572    @staticmethod573    def _get_tags(runs: Sequence[Run]) -> list[str]:574        """Extract tags from runs."""575        # Add tags to support filtering576        tags = []577        for run in runs:578            tags.append("run:" + str(run.id))579            if getattr(run, "session_id", None):580                tags.append("experiment:" + str(run.session_id))581        return tags582 583    def _format_results(584        self,585        result: Union[dict, list, ComparisonEvaluationResult],586        source_run_id: uuid.UUID,587        runs: Sequence[Run],588    ) -> ComparisonEvaluationResult:589        if isinstance(result, ComparisonEvaluationResult):590            if not result.source_run_id:591                result.source_run_id = source_run_id592            return result593        elif isinstance(result, list):594            result = {595                "scores": {run.id: score for run, score in zip(runs, result)},596                "key": self._name,597                "source_run_id": source_run_id,598            }599        elif isinstance(result, dict):600            if "key" not in result:601                result["key"] = self._name602        else:603            msg = (604                "Expected 'dict', 'list' or 'ComparisonEvaluationResult' result "605                f"object. Received: {result=}"606            )607            raise ValueError(msg)608        try:609            return ComparisonEvaluationResult(610                **{"source_run_id": source_run_id, **result}611            )612        except ValidationError as e:613            raise ValueError(614                f"Expected a dictionary with a 'key' and dictionary of scores mapping"615                "run IDs to numeric scores, or ComparisonEvaluationResult object,"616                f" got {result}"617            ) from e618 619 620def comparison_evaluator(621    func: Callable[622        [Sequence[Run], Optional[Example]],623        Union[_COMPARISON_OUTPUT, Awaitable[_COMPARISON_OUTPUT]],624    ],625) -> DynamicComparisonRunEvaluator:626    """Create a comaprison evaluator from a function."""627    return DynamicComparisonRunEvaluator(func)628 629 630def _normalize_evaluator_func(631    func: Callable,632) -> tuple[633    Union[634        Callable[[Run, Optional[Example]], _RUNNABLE_OUTPUT],635        Callable[[Run, Optional[Example]], Awaitable[_RUNNABLE_OUTPUT]],636    ],637    Optional[Callable[..., dict]],638]:639    supported_args = (640        "run",641        "example",642        "inputs",643        "outputs",644        "reference_outputs",645        "attachments",646    )647    sig = inspect.signature(func)648    all_args = [pname for pname, p in sig.parameters.items() if p.kind != p.VAR_KEYWORD]649    args_with_defaults = [650        pname651        for pname, p in sig.parameters.items()652        if p.default is not inspect.Parameter.empty653    ]654    if not all_args or (655        not all(656            pname in supported_args or pname in args_with_defaults for pname in all_args657        )658        and len([a for a in all_args if a not in args_with_defaults]) != 2659    ):660        msg = (661            f"Invalid evaluator function. Must have at least one "662            f"argument. Supported arguments are {supported_args}. Please "663            f"see https://docs.smith.langchain.com/evaluation/how_to_guides/evaluation/evaluate_llm_application#use-custom-evaluators"664            # noqa: E501665        )666        raise ValueError(msg)667    # For backwards compatibility we assume custom arg names are Run and Example668    # types, respectively.669    elif not all(670        pname in supported_args or pname in args_with_defaults for pname in all_args671    ) or all_args == [672        "run",673        "example",674    ]:675        return func, None676    else:677        if inspect.iscoroutinefunction(func):678 679            def _prepare_inputs(680                run: Run, example: Optional[Example]681            ) -> tuple[list, dict, dict]:682                arg_map = {683                    "run": run,684                    "example": example,685                    "inputs": example.inputs if example else {},686                    "outputs": run.outputs or {},687                    "attachments": example.attachments or {} if example else {},688                    "reference_outputs": example.outputs or {} if example else {},689                }690                kwargs = {}691                args = []692                traced_inputs = {}693                for param_name, param in sig.parameters.items():694                    # Could have params with defaults that are not in the arg map695                    if param_name in arg_map:696                        if param.kind in (697                            param.POSITIONAL_OR_KEYWORD,698                            param.POSITIONAL_ONLY,699                        ):700                            args.append(arg_map[param_name])701                        else:702                            kwargs[param_name] = arg_map[param_name]703                        traced_inputs[param_name] = (704                            _maxsize_repr(arg_map[param_name])705                            if param_name in ("run", "example")706                            else arg_map[param_name]707                        )708                return args, kwargs, traced_inputs709 710            async def awrapper(711                run: Run, example: Optional[Example]712            ) -> _RUNNABLE_OUTPUT:713                (args, kwargs, _) = _prepare_inputs(run, example)714                return await func(*args, **kwargs)715 716            awrapper.__name__ = (717                getattr(func, "__name__")718                if hasattr(func, "__name__")719                else awrapper.__name__720            )721            return (awrapper, _prepare_inputs)  # type: ignore[return-value]722 723        else:724 725            def _prepare_inputs(726                run: Run, example: Optional[Example]727            ) -> tuple[list, dict, dict]:728                arg_map = {729                    "run": run,730                    "example": example,731                    "inputs": example.inputs if example else {},732                    "outputs": run.outputs or {},733                    "attachments": example.attachments or {} if example else {},734                    "reference_outputs": example.outputs or {} if example else {},735                }736                kwargs = {}737                args = []738                traced_inputs = {}739                for param_name, param in sig.parameters.items():740                    # Could have params with defaults that are not in the arg map741                    if param_name in arg_map:742                        if param.kind in (743                            param.POSITIONAL_OR_KEYWORD,744                            param.POSITIONAL_ONLY,745                        ):746                            args.append(arg_map[param_name])747                        else:748                            kwargs[param_name] = arg_map[param_name]749                        traced_inputs[param_name] = (750                            _maxsize_repr(arg_map[param_name])751                            if param_name in ("run", "example")752                            else arg_map[param_name]753                        )754                return args, kwargs, traced_inputs755 756            def wrapper(run: Run, example: Optional[Example]) -> _RUNNABLE_OUTPUT:757                (args, kwargs, _) = _prepare_inputs(run, example)758                return func(*args, **kwargs)759 760            wrapper.__name__ = (761                getattr(func, "__name__")762                if hasattr(func, "__name__")763                else wrapper.__name__764            )765            return (wrapper, _prepare_inputs)  # type: ignore[return-value]766 767 768def _normalize_comparison_evaluator_func(769    func: Callable,770) -> tuple[771    Union[772        Callable[[Sequence[Run], Optional[Example]], _COMPARISON_OUTPUT],773        Callable[[Sequence[Run], Optional[Example]], Awaitable[_COMPARISON_OUTPUT]],774    ],775    Optional[Callable[..., dict]],776]:777    supported_args = ("runs", "example", "inputs", "outputs", "reference_outputs")778    sig = inspect.signature(func)779    all_args = [pname for pname, p in sig.parameters.items() if p.kind != p.VAR_KEYWORD]780    args_with_defaults = [781        pname782        for pname, p in sig.parameters.items()783        if p.default is not inspect.Parameter.empty784    ]785    if not all_args or (786        not all(787            pname in supported_args or pname in args_with_defaults for pname in all_args788        )789        and len([a for a in all_args if a not in args_with_defaults]) != 2790    ):791        msg = (792            f"Invalid evaluator function. Must have at least one "793            f"argument. Supported arguments are {supported_args}. Please "794            f"see https://docs.smith.langchain.com/evaluation/how_to_guides/evaluation/evaluate_llm_application#use-custom-evaluators"795            # noqa: E501796        )797        raise ValueError(msg)798    # For backwards compatibility we assume custom arg names are List[Run] and799    # List[Example] types, respectively.800    elif not all(801        pname in supported_args or pname in args_with_defaults for pname in all_args802    ) or all_args == [803        "runs",804        "example",805    ]:806        return func, None807    else:808        if inspect.iscoroutinefunction(func):809 810            def _prepare_inputs(811                runs: Sequence[Run], example: Optional[Example]812            ) -> tuple[list, dict, dict]:813                arg_map = {814                    "runs": runs,815                    "example": example,816                    "inputs": example.inputs if example else {},817                    "outputs": [run.outputs or {} for run in runs],818                    "reference_outputs": example.outputs or {} if example else {},819                }820                kwargs = {}821                args = []822                traced_inputs = {}823                for param_name, param in sig.parameters.items():824                    # Could have params with defaults that are not in the arg map825                    if param_name in arg_map:826                        if param.kind in (827                            param.POSITIONAL_OR_KEYWORD,828                            param.POSITIONAL_ONLY,829                        ):830                            args.append(arg_map[param_name])831                        else:832                            kwargs[param_name] = arg_map[param_name]833                        traced_inputs[param_name] = (834                            _maxsize_repr(arg_map[param_name])835                            if param_name in ("runs", "example")836                            else arg_map[param_name]837                        )838                return args, kwargs, traced_inputs839 840            async def awrapper(841                runs: Sequence[Run], example: Optional[Example]842            ) -> _COMPARISON_OUTPUT:843                (args, kwargs, _) = _prepare_inputs(runs, example)844                return await func(*args, **kwargs)845 846            awrapper.__name__ = (847                getattr(func, "__name__")848                if hasattr(func, "__name__")849                else awrapper.__name__850            )851            return awrapper, _prepare_inputs  # type: ignore[return-value]852 853        else:854 855            def _prepare_inputs(856                runs: Sequence[Run], example: Optional[Example]857            ) -> tuple[list, dict, dict]:858                arg_map = {859                    "runs": runs,860                    "example": example,861                    "inputs": example.inputs if example else {},862                    "outputs": [run.outputs or {} for run in runs],863                    "reference_outputs": example.outputs or {} if example else {},864                }865                kwargs = {}866                args = []867                traced_inputs = {}868                for param_name, param in sig.parameters.items():869                    # Could have params with defaults that are not in the arg map870                    if param_name in arg_map:871                        if param.kind in (872                            param.POSITIONAL_OR_KEYWORD,873                            param.POSITIONAL_ONLY,874                        ):875                            args.append(arg_map[param_name])876                        else:877                            kwargs[param_name] = arg_map[param_name]878                        traced_inputs[param_name] = (879                            _maxsize_repr(arg_map[param_name])880                            if param_name in ("runs", "example")881                            else arg_map[param_name]882                        )883                return args, kwargs, traced_inputs884 885            def wrapper(886                runs: Sequence[Run], example: Optional[Example]887            ) -> _COMPARISON_OUTPUT:888                (args, kwargs, _) = _prepare_inputs(runs, example)889                return func(*args, **kwargs)890 891            wrapper.__name__ = (892                getattr(func, "__name__")893                if hasattr(func, "__name__")894                else wrapper.__name__895            )896            return wrapper, _prepare_inputs  # type: ignore[return-value]897 898 899def _format_evaluator_result(900    result: Union[EvaluationResults, dict, str, int, bool, float, list],901) -> Union[EvaluationResults, dict]:902    if isinstance(result, (bool, float, int)):903        result = {"score": result}904    elif not result:905        raise ValueError(906            f"Expected a non-empty dict, str, bool, int, float, list, "907            f"EvaluationResult, or EvaluationResults. Got {result}"908        )909    elif isinstance(result, list):910        if not all(isinstance(x, dict) for x in result):911            raise ValueError(912                f"Expected a list of dicts or EvaluationResults. Received {result}."913            )914        result = {"results": result}  # type: ignore[misc]915    elif isinstance(result, str):916        result = {"value": result}917    elif isinstance(result, dict):918        pass919    else:920        raise ValueError(921            f"Expected a dict, str, bool, int, float, list, EvaluationResult, or "922            f"EvaluationResults. Got {result}"923        )924    return result925 926 927SUMMARY_EVALUATOR_T = Union[928    Callable[929        [Sequence[schemas.Run], Sequence[schemas.Example]],930        Union[EvaluationResult, EvaluationResults],931    ],932    Callable[933        [list[schemas.Run], list[schemas.Example]],934        Union[EvaluationResult, EvaluationResults],935    ],936]937 938 939def _normalize_summary_evaluator(func: Callable) -> SUMMARY_EVALUATOR_T:940    supported_args = ("runs", "examples", "inputs", "outputs", "reference_outputs")941    sig = inspect.signature(func)942    all_args = [pname for pname, p in sig.parameters.items()]943    args_with_defaults = [944        pname945        for pname, p in sig.parameters.items()946        if p.default is not inspect.Parameter.empty947    ]948    if not all_args or (949        not all(950            pname in supported_args or pname in args_with_defaults for pname in all_args951        )952        and len([a for a in all_args if a not in args_with_defaults]) != 2953    ):954        msg = (955            f"Invalid evaluator function. Must have at least one "956            f"argument. Supported arguments are {supported_args}."957        )958        if all_args:959            msg += f" Received arguments {all_args}."960        raise ValueError(msg)961    # For backwards compatibility we assume custom arg names are Sequence[Run] and962    # Sequence[Example] types, respectively.963    elif not all(pname in supported_args for pname in all_args) or all_args == [964        "runs",965        "examples",966    ]:967        return func968    else:969 970        def wrapper(971            runs: Sequence[schemas.Run], examples: Sequence[schemas.Example]972        ) -> Union[EvaluationResult, EvaluationResults]:973            arg_map = {974                "runs": runs,975                "examples": examples,976                "inputs": [example.inputs for example in examples],977                "outputs": [run.outputs or {} for run in runs],978                "reference_outputs": [example.outputs or {} for example in examples],979            }980            kwargs = {}981            args = []982            for param_name, param in sig.parameters.items():983                # Could have params with defaults that are not in the arg map984                if param_name in arg_map:985                    if param.kind in (986                        param.POSITIONAL_OR_KEYWORD,987                        param.POSITIONAL_ONLY,988                    ):989                        args.append(arg_map[param_name])990                    else:991                        kwargs[param_name] = arg_map[param_name]992 993            result = func(*args, **kwargs)994            if isinstance(result, EvaluationResult):995                return result996            return _format_evaluator_result(result)  # type: ignore997 998        wrapper.__name__ = (999            getattr(func, "__name__") if hasattr(func, "__name__") else wrapper.__name__1000        )1001        return wrapper  # type: ignore[return-value]1002 
codekingpro/portable-devtools · Team Ai