Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
_runner.py2315 linesDownload Raw Back to evaluation
1"""V2 Evaluation Interface."""2 3from __future__ import annotations4 5import ast6import collections7import concurrent.futures as cf8import functools9import inspect10import io11import itertools12import logging13import pathlib14import queue15import random16import textwrap17import threading18import uuid19from collections.abc import Awaitable, Generator, Iterable, Iterator, Sequence20from contextvars import copy_context21from typing import (22    TYPE_CHECKING,23    Any,24    Callable,25    Literal,26    Optional,27    TypeVar,28    Union,29    cast,30)31 32from typing_extensions import TypedDict, overload33 34import langsmith35from langsmith import env as ls_env36from langsmith import run_helpers as rh37from langsmith import run_trees as rt38from langsmith import schemas39from langsmith import utils as ls_utils40from langsmith._internal._beta_decorator import _warn_once41from langsmith.evaluation.evaluator import (42    SUMMARY_EVALUATOR_T,43    ComparisonEvaluationResult,44    DynamicComparisonRunEvaluator,45    DynamicRunEvaluator,46    EvaluationResult,47    EvaluationResults,48    RunEvaluator,49    _normalize_summary_evaluator,50    comparison_evaluator,51    run_evaluator,52)53 54# Python 3.14+ removes ast.Str in favor of ast.Constant55_AST_STR_TYPES: tuple = (56    (ast.Str, ast.Constant) if hasattr(ast, "Str") else (ast.Constant,)57)58 59 60def _get_str_value(node: ast.expr) -> str:61    """Get string value from ast.Str or ast.Constant."""62    return node.value if isinstance(node, ast.Constant) else node.s  # type: ignore[return-value,union-attr,attr-defined]63 64 65if TYPE_CHECKING:66    import pandas as pd67    from langchain_core.runnables import Runnable68 69    DataFrame = pd.DataFrame70else:71    DataFrame = Any72logger = logging.getLogger(__name__)73 74TARGET_T = Union[Callable[[dict], dict], Callable[[dict, dict], dict]]75# Data format: dataset-name, dataset_id, or examples76DATA_T = Union[str, uuid.UUID, Iterable[schemas.Example], schemas.Dataset]77# Summary evaluator runs over the whole dataset78# and reports aggregate metric(s)79# Row-level evaluator80EVALUATOR_T = Union[81    RunEvaluator,82    Callable[83        [schemas.Run, Optional[schemas.Example]],84        Union[EvaluationResult, EvaluationResults],85    ],86    Callable[..., Union[dict, EvaluationResults, EvaluationResult]],87]88AEVALUATOR_T = Union[89    Callable[90        [schemas.Run, Optional[schemas.Example]],91        Awaitable[Union[EvaluationResult, EvaluationResults]],92    ],93]94EXPERIMENT_T = Union[str, uuid.UUID, schemas.TracerSession]95 96 97@overload98def evaluate(99    target: Union[TARGET_T, Runnable, EXPERIMENT_T],100    /,101    data: Optional[DATA_T] = None,102    evaluators: Optional[Sequence[EVALUATOR_T]] = None,103    summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,104    metadata: Optional[dict] = None,105    experiment_prefix: Optional[str] = None,106    description: Optional[str] = None,107    max_concurrency: Optional[int] = 0,108    num_repetitions: int = 1,109    client: Optional[langsmith.Client] = None,110    blocking: bool = True,111    experiment: Optional[EXPERIMENT_T] = None,112    upload_results: bool = True,113    **kwargs: Any,114) -> ExperimentResults: ...115 116 117@overload118def evaluate(119    target: Union[tuple[EXPERIMENT_T, EXPERIMENT_T]],120    /,121    data: Optional[DATA_T] = None,122    evaluators: Optional[Sequence[COMPARATIVE_EVALUATOR_T]] = None,123    summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,124    metadata: Optional[dict] = None,125    experiment_prefix: Optional[str] = None,126    description: Optional[str] = None,127    max_concurrency: Optional[int] = 0,128    num_repetitions: int = 1,129    client: Optional[langsmith.Client] = None,130    blocking: bool = True,131    experiment: Optional[EXPERIMENT_T] = None,132    upload_results: bool = True,133    **kwargs: Any,134) -> ComparativeExperimentResults: ...135 136 137def evaluate(138    target: Union[TARGET_T, Runnable, EXPERIMENT_T, tuple[EXPERIMENT_T, EXPERIMENT_T]],139    /,140    data: Optional[DATA_T] = None,141    evaluators: Optional[142        Union[Sequence[EVALUATOR_T], Sequence[COMPARATIVE_EVALUATOR_T]]143    ] = None,144    summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,145    metadata: Optional[dict] = None,146    experiment_prefix: Optional[str] = None,147    description: Optional[str] = None,148    max_concurrency: Optional[int] = 0,149    num_repetitions: int = 1,150    client: Optional[langsmith.Client] = None,151    blocking: bool = True,152    experiment: Optional[EXPERIMENT_T] = None,153    upload_results: bool = True,154    error_handling: Literal["log", "ignore"] = "log",155    **kwargs: Any,156) -> Union[ExperimentResults, ComparativeExperimentResults]:157    r"""Evaluate a target system on a given dataset.158 159    Args:160        target (TARGET_T | Runnable | EXPERIMENT_T | Tuple[EXPERIMENT_T, EXPERIMENT_T]):161            The target system or experiment(s) to evaluate.162 163            Can be a function that takes a dict and returns a `dict`, a langchain `Runnable`, an164            existing experiment ID, or a two-tuple of experiment IDs.165        data (DATA_T): The dataset to evaluate on.166 167            Can be a dataset name, a list of examples, or a generator of examples.168        evaluators (Sequence[EVALUATOR_T] | Sequence[COMPARATIVE_EVALUATOR_T] | None):169            A list of evaluators to run on each example. The evaluator signature170            depends on the target type.171        summary_evaluators (Sequence[SUMMARY_EVALUATOR_T] | None): A list of summary172            evaluators to run on the entire dataset.173 174            Should not be specified if comparing two existing experiments.175        metadata (dict | None): Metadata to attach to the experiment.176        experiment_prefix (str | None): A prefix to provide for your experiment name.177        description (str | None): A free-form text description for the experiment.178        max_concurrency (int | None): The maximum number of concurrent179            evaluations to run.180 181            If `None` then no limit is set. If `0` then no concurrency.182        client (langsmith.Client | None): The LangSmith client to use.183        blocking (bool): Whether to block until the evaluation is complete.184        num_repetitions (int): The number of times to run the evaluation.185            Each item in the dataset will be run and evaluated this many times.186        experiment (schemas.TracerSession | None): An existing experiment to187            extend.188 189            If provided, `experiment_prefix` is ignored.190 191            For advanced usage only. Should not be specified if target is an existing192            experiment or two-tuple fo experiments.193        error_handling (str, default="log"): How to handle individual run errors.194 195            `'log'` will trace the runs with the error message as part of the196            experiment, `'ignore'` will not count the run as part of the experiment at197            all.198 199    Returns:200        ExperimentResults: If target is a function, `Runnable`, or existing experiment.201        ComparativeExperimentResults: If target is a two-tuple of existing experiments.202 203    Examples:204        Prepare the dataset:205 206        >>> from typing import Sequence207        >>> from langsmith import Client208        >>> from langsmith.evaluation import evaluate209        >>> from langsmith.schemas import Example, Run210        >>> client = Client()211        >>> dataset = client.clone_public_dataset(212        ...     "https://smith.langchain.com/public/419dcab2-1d66-4b94-8901-0357ead390df/d"213        ... )214        >>> dataset_name = "Evaluate Examples"215 216        Basic usage:217 218        >>> def accuracy(run: Run, example: Example):219        ...     # Row-level evaluator for accuracy.220        ...     pred = run.outputs["output"]221        ...     expected = example.outputs["answer"]222        ...     return {"score": expected.lower() == pred.lower()}223        >>> def precision(runs: Sequence[Run], examples: Sequence[Example]):224        ...     # Experiment-level evaluator for precision.225        ...     # TP / (TP + FP)226        ...     predictions = [run.outputs["output"].lower() for run in runs]227        ...     expected = [example.outputs["answer"].lower() for example in examples]228        ...     # yes and no are the only possible answers229        ...     tp = sum([p == e for p, e in zip(predictions, expected) if p == "yes"])230        ...     fp = sum([p == "yes" and e == "no" for p, e in zip(predictions, expected)])231        ...     return {"score": tp / (tp + fp)}232        >>> def predict(inputs: dict) -> dict:233        ...     # This can be any function or just an API call to your app.234        ...     return {"output": "Yes"}235        >>> results = evaluate(236        ...     predict,237        ...     data=dataset_name,238        ...     evaluators=[accuracy],239        ...     summary_evaluators=[precision],240        ...     experiment_prefix="My Experiment",241        ...     description="Evaluating the accuracy of a simple prediction model.",242        ...     metadata={243        ...         "my-prompt-version": "abcd-1234",244        ...     },245        ... )  # doctest: +ELLIPSIS246        View the evaluation results for experiment:...247 248        Evaluating over only a subset of the examples249 250        >>> experiment_name = results.experiment_name251        >>> examples = client.list_examples(dataset_name=dataset_name, limit=5)252        >>> results = evaluate(253        ...     predict,254        ...     data=examples,255        ...     evaluators=[accuracy],256        ...     summary_evaluators=[precision],257        ...     experiment_prefix="My Experiment",258        ...     description="Just testing a subset synchronously.",259        ... )  # doctest: +ELLIPSIS260        View the evaluation results for experiment:...261 262        Streaming each prediction to more easily + eagerly debug.263 264        >>> results = evaluate(265        ...     predict,266        ...     data=dataset_name,267        ...     evaluators=[accuracy],268        ...     summary_evaluators=[precision],269        ...     description="I don't even have to block!",270        ...     blocking=False,271        ... )  # doctest: +ELLIPSIS272        View the evaluation results for experiment:...273        >>> for i, result in enumerate(results):  # doctest: +ELLIPSIS274        ...     pass275 276 277 278        Evaluating a LangChain object:279 280        >>> from langchain_core.runnables import chain as as_runnable281        >>> @as_runnable282        ... def nested_predict(inputs):283        ...     return {"output": "Yes"}284        >>> @as_runnable285        ... def lc_predict(inputs):286        ...     return nested_predict.invoke(inputs)287        >>> results = evaluate(288        ...     lc_predict.invoke,289        ...     data=dataset_name,290        ...     evaluators=[accuracy],291        ...     description="This time we're evaluating a LangChain object.",292        ...     summary_evaluators=[precision],293        ... )  # doctest: +ELLIPSIS294        View the evaluation results for experiment:...295 296    !!! warning "Behavior changed in `langsmith` 0.2.0"297 298        'max_concurrency' default updated from None (no limit on concurrency)299        to 0 (no concurrency at all).300    """  # noqa: E501301    if isinstance(target, (str, uuid.UUID, schemas.TracerSession)):302        invalid_args = {303            "num_repetitions": num_repetitions > 1,304            "experiment": bool(experiment),305            "upload_results": not upload_results,306            "experiment_prefix": bool(experiment_prefix),307            "data": bool(data),308        }309        if any(invalid_args.values()):310            msg = (311                f"Received invalid arguments. "312                f"{tuple(k for k, v in invalid_args.items() if v)} should not be "313                f"specified when target is an existing experiment."314            )315            raise ValueError(msg)316        target_id = target if isinstance(target, (str, uuid.UUID)) else target.id317        logger.debug(f"Running evaluation over existing experiment {target_id}...")318        return evaluate_existing(319            target,320            evaluators=cast(Optional[Sequence[EVALUATOR_T]], evaluators),321            summary_evaluators=summary_evaluators,322            metadata=metadata,323            max_concurrency=max_concurrency,324            client=client,325            blocking=blocking,326            **kwargs,327        )328    elif isinstance(target, (list, tuple)):329        invalid_args = {330            "num_repetitions": num_repetitions > 1,331            "experiment": bool(experiment),332            "upload_results": not upload_results,333            "summary_evaluators": bool(summary_evaluators),334            "data": bool(data),335        }336        if len(target) != 2 or not all(337            isinstance(t, (str, uuid.UUID, schemas.TracerSession)) for t in target338        ):339            msg = (340                "Received invalid target. If a tuple is specified it must have length "341                "2 and each element should by the ID or schemas.TracerSession of an "342                f"existing experiment. Received {target=}"343            )344            raise ValueError(msg)345        elif any(invalid_args.values()):346            msg = (347                f"Received invalid arguments. "348                f"{tuple(k for k, v in invalid_args.items() if v)} should not be "349                f"specified when target is two existing experiments."350            )351            raise ValueError(msg)352        if max_concurrency is not None:353            kwargs["max_concurrency"] = max_concurrency354        target_ids = [t if isinstance(t, (str, uuid.UUID)) else t.id for t in target]355        logger.debug(356            f"Running pairwise evaluation over existing experiments {target_ids}..."357        )358        return evaluate_comparative(359            target,360            evaluators=cast(Sequence[COMPARATIVE_EVALUATOR_T], evaluators or ()),361            experiment_prefix=experiment_prefix,362            description=description,363            client=client,364            metadata=metadata,365            **kwargs,366        )367    elif kwargs:368        msg = (369            f"Received unsupported arguments {kwargs}. These arguments are not "370            f"supported when creating a new experiment."371        )372        raise ValueError(msg)373    elif not data:374        msg = "Must specify 'data' when running evaluations over a target function."375        raise ValueError(msg)376    elif callable(target) and rh.is_async(target):377        msg = (378            "Async functions are not supported by `evaluate`. "379            "Please use `aevaluate` instead:\n\n"380            "from langsmith import aevaluate\n\n"381            "await aevaluate(\n"382            "    async_target_function,\n"383            "    data=data,\n"384            "    evaluators=evaluators,\n"385            "    # ... other parameters\n"386            ")"387        )388        raise ValueError(msg)389    elif experiment and experiment_prefix:390        msg = (391            "Expected at most one of 'experiment' or 'experiment_prefix',"392            " but both were provided. "393            f"Got: experiment={experiment}, experiment_prefix={experiment_prefix}"394        )395        raise ValueError(msg)396    else:397        if not upload_results:398            _warn_once("'upload_results' parameter is in beta.")399        logger.debug(f"Running evaluation over target system {target}...")400        return _evaluate(401            target,402            data=data,403            evaluators=cast(Optional[Sequence[EVALUATOR_T]], evaluators),404            summary_evaluators=summary_evaluators,405            metadata=metadata,406            experiment_prefix=experiment_prefix,407            description=description,408            max_concurrency=max_concurrency,409            num_repetitions=num_repetitions,410            client=client,411            blocking=blocking,412            experiment=experiment,413            upload_results=upload_results,414            error_handling=error_handling,415        )416 417 418def evaluate_existing(419    experiment: Union[str, uuid.UUID, schemas.TracerSession],420    /,421    evaluators: Optional[Sequence[EVALUATOR_T]] = None,422    summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,423    metadata: Optional[dict] = None,424    max_concurrency: Optional[int] = 0,425    client: Optional[langsmith.Client] = None,426    load_nested: bool = False,427    blocking: bool = True,428) -> ExperimentResults:429    r"""Evaluate existing experiment runs.430 431    Args:432        experiment (Union[str, uuid.UUID]): The identifier of the experiment to evaluate.433        evaluators (Optional[Sequence[EVALUATOR_T]]): Optional sequence of evaluators to use for individual run evaluation.434        summary_evaluators (Optional[Sequence[SUMMARY_EVALUATOR_T]]): Optional sequence of evaluators435            to apply over the entire dataset.436        metadata (Optional[dict]): Optional metadata to include in the evaluation results.437        max_concurrency (int | None): The maximum number of concurrent438            evaluations to run.439 440            If `None` then no limit is set. If `0` then no concurrency.441        client (Optional[langsmith.Client]): Optional Langsmith client to use for evaluation.442        load_nested: Whether to load all child runs for the experiment.443 444            Default is to only load the top-level root runs.445        blocking (bool): Whether to block until evaluation is complete.446 447    Returns:448        The evaluation results.449 450    Environment:451        - `LANGSMITH_TEST_CACHE`: If set, API calls will be cached to disk to save time and452            cost during testing.453 454            Recommended to commit the cache files to your repository for faster CI/CD runs.455 456            Requires the `'langsmith[vcr]'` package to be installed.457 458    Examples:459        Define your evaluators460 461        >>> from typing import Sequence462        >>> from langsmith.schemas import Example, Run463        >>> def accuracy(run: Run, example: Example):464        ...     # Row-level evaluator for accuracy.465        ...     pred = run.outputs["output"]466        ...     expected = example.outputs["answer"]467        ...     return {"score": expected.lower() == pred.lower()}468        >>> def precision(runs: Sequence[Run], examples: Sequence[Example]):469        ...     # Experiment-level evaluator for precision.470        ...     # TP / (TP + FP)471        ...     predictions = [run.outputs["output"].lower() for run in runs]472        ...     expected = [example.outputs["answer"].lower() for example in examples]473        ...     # yes and no are the only possible answers474        ...     tp = sum([p == e for p, e in zip(predictions, expected) if p == "yes"])475        ...     fp = sum([p == "yes" and e == "no" for p, e in zip(predictions, expected)])476        ...     return {"score": tp / (tp + fp)}477 478        Load the experiment and run the evaluation.479 480        >>> import uuid481        >>> from langsmith import Client482        >>> from langsmith.evaluation import evaluate, evaluate_existing483        >>> client = Client()484        >>> dataset_name = "__doctest_evaluate_existing_" + uuid.uuid4().hex[:8]485        >>> dataset = client.create_dataset(dataset_name)486        >>> example = client.create_example(487        ...     inputs={"question": "What is 2+2?"},488        ...     outputs={"answer": "4"},489        ...     dataset_id=dataset.id,490        ... )491        >>> def predict(inputs: dict) -> dict:492        ...     return {"output": "4"}493        >>> # First run inference on the dataset494        ... results = evaluate(495        ...     predict, data=dataset_name, experiment_prefix="doctest_experiment"496        ... )  # doctest: +ELLIPSIS497        View the evaluation results for experiment:...498        >>> experiment_id = results.experiment_name499        >>> # Wait for the experiment to be fully processed and check if we have results500        >>> len(results) > 0501        True502        >>> import time503        >>> time.sleep(5)  # Wait longer for runs to be indexed504        >>> results = evaluate_existing(505        ...     experiment_id,506        ...     evaluators=[accuracy],507        ...     summary_evaluators=[precision],508        ... )  # doctest: +ELLIPSIS509        View the evaluation results for experiment:...510        >>> client.delete_dataset(dataset_id=dataset.id)511    """  # noqa: E501512    client = client or rt.get_cached_client(timeout_ms=(20_000, 90_001))513    project = _load_experiment(experiment, client)514    runs = _load_traces(experiment, client, load_nested=load_nested)515    data_map = _load_examples_map(client, project)516    data = [data_map[cast(uuid.UUID, run.reference_example_id)] for run in runs]517    return _evaluate(518        runs,519        data=data,520        evaluators=evaluators,521        summary_evaluators=summary_evaluators,522        metadata=metadata,523        max_concurrency=max_concurrency,524        client=client,525        blocking=blocking,526        experiment=project,527    )528 529 530class ExperimentResultRow(TypedDict):531    run: schemas.Run532    example: schemas.Example533    evaluation_results: EvaluationResults534 535 536class ExperimentResults:537    """Represents the results of an evaluate() call.538 539    This class provides an iterator interface to iterate over the experiment results540    as they become available. It also provides methods to access the experiment name,541    the number of results, and to wait for the results to be processed.542 543    Methods:544        experiment_name() -> str: Returns the name of the experiment.545        wait() -> None: Waits for the experiment data to be processed.546    """547 548    def __init__(self, experiment_manager: _ExperimentManager, blocking: bool = True):549        self._manager = experiment_manager550        self._results: list[ExperimentResultRow] = []551        self._queue: queue.Queue[ExperimentResultRow] = queue.Queue()552        self._processing_complete = threading.Event()553        self._processing_error: Optional[BaseException] = None554        if not blocking:555            self._thread: Optional[threading.Thread] = threading.Thread(556                target=self._process_data557            )558            self._thread.start()559        else:560            self._thread = None561            self._process_data()562 563    @property564    def experiment_name(self) -> str:565        return self._manager.experiment_name566 567    @property568    def experiment_id(self) -> uuid.UUID:569        """The ID of the experiment."""570        return self._manager._get_experiment().id571 572    @property573    def url(self) -> Optional[str]:574        """The URL of the experiment in the LangSmith UI."""575        experiment = self._manager._get_experiment()576        if experiment.url:577            project_url = experiment.url.split("?")[0]578            base_url = project_url.split("/projects/p/")[0]579            return (580                f"{base_url}/datasets/{self._manager.dataset_id}/compare?"581                f"selectedSessions={experiment.id}"582            )583        return None584 585    def get_dataset_id(self) -> str:586        """Get the ID of the dataset associated with this experiment."""587        return self._manager.dataset_id588 589    @property590    def comparison_url(self) -> Optional[str]:591        """The URL to the comparison view for this experiment."""592        return self.url593 594    def __iter__(self) -> Iterator[ExperimentResultRow]:595        ix = 0596        while (597            not self._processing_complete.is_set()598            or not self._queue.empty()599            or ix < len(self._results)600        ):601            try:602                if ix < len(self._results):603                    yield self._results[ix]604                    ix += 1605                else:606                    self._queue.get(block=True, timeout=0.1)607            except queue.Empty:608                if self._processing_error is not None:609                    raise self._processing_error610                continue611        if self._processing_error is not None:612            raise self._processing_error613 614    def _process_data(self) -> None:615        tqdm = _load_tqdm()616        try:617            results = self._manager.get_results()618            for item in tqdm(results):619                self._queue.put(item)620                self._results.append(item)621 622            summary_scores = self._manager.get_summary_scores()623            self._summary_results = summary_scores624        except BaseException as e:625            self._processing_error = e626        finally:627            self._processing_complete.set()628 629    def __len__(self) -> int:630        return len(self._results)631 632    def to_pandas(633        self, start: Optional[int] = 0, end: Optional[int] = None634    ) -> DataFrame:635        return _to_pandas(self._results, start=start, end=end)636 637    def _repr_html_(self) -> str:638        import importlib.util639 640        if self._results and importlib.util.find_spec("pandas"):641            df = self.to_pandas()642            return df._repr_html_()  # type: ignore[operator]643        else:644            return self.__repr__()645 646    def __repr__(self) -> str:647        return f"<ExperimentResults {self.experiment_name}>"648 649    def wait(self) -> None:650        """Wait for the evaluation runner to complete.651 652        This method blocks the current thread until the evaluation runner has653        finished its execution.654        """655        if self._thread:656            self._thread.join()657        if self._processing_error is not None:658            raise self._processing_error659 660 661## Public API for Comparison Experiments662 663# Row-level evaluator664COMPARATIVE_EVALUATOR_T = Callable[665    [Sequence[schemas.Run], Optional[schemas.Example]],666    Union[667        Union[ComparisonEvaluationResult, dict],668        Awaitable[Union[ComparisonEvaluationResult, dict]],669    ],670]671 672 673def evaluate_comparative(674    experiments: tuple[EXPERIMENT_T, EXPERIMENT_T],675    /,676    evaluators: Sequence[COMPARATIVE_EVALUATOR_T],677    experiment_prefix: Optional[str] = None,678    description: Optional[str] = None,679    max_concurrency: int = 5,680    client: Optional[langsmith.Client] = None,681    metadata: Optional[dict] = None,682    load_nested: bool = False,683    randomize_order: bool = False,684) -> ComparativeExperimentResults:685    r"""Evaluate existing experiment runs against each other.686 687    This lets you use pairwise preference scoring to generate more688    reliable feedback in your experiments.689 690    Args:691        experiments (Tuple[Union[str, uuid.UUID], Union[str, uuid.UUID]]):692            The identifiers of the experiments to compare.693        evaluators (Sequence[COMPARATIVE_EVALUATOR_T]):694            A list of evaluators to run on each example.695        experiment_prefix (Optional[str]): A prefix to provide for your experiment name.696        description (Optional[str]): A free-form text description for the experiment.697        max_concurrency (int): The maximum number of concurrent evaluations to run.698        client (Optional[langsmith.Client]): The LangSmith client to use.699        metadata (Optional[dict]): Metadata to attach to the experiment.700        load_nested (bool): Whether to load all child runs for the experiment.701 702            Default is to only load the top-level root runs.703        randomize_order (bool): Whether to randomize the order of the outputs for each evaluation.704 705    Returns:706        The results of the comparative evaluation.707 708    Examples:709        Suppose you want to compare two prompts to see which one is more effective.710        You would first prepare your dataset:711 712        >>> from typing import Sequence713        >>> from langsmith import Client714        >>> from langsmith.evaluation import evaluate715        >>> from langsmith.schemas import Example, Run716        >>> client = Client()717        >>> dataset = client.clone_public_dataset(718        ...     "https://smith.langchain.com/public/419dcab2-1d66-4b94-8901-0357ead390df/d"719        ... )720        >>> dataset_name = "Evaluate Examples"721 722        Then you would run your different prompts:723        >>> import functools724        >>> import openai725        >>> from langsmith.evaluation import evaluate726        >>> from langsmith.wrappers import wrap_openai727        >>> oai_client = openai.Client()728        >>> wrapped_client = wrap_openai(oai_client)729        >>> prompt_1 = "You are a helpful assistant."730        >>> prompt_2 = "You are an exceedingly helpful assistant."731        >>> def predict(inputs: dict, prompt: str) -> dict:732        ...     completion = wrapped_client.chat.completions.create(733        ...         model="gpt-4o-mini",734        ...         messages=[735        ...             {"role": "system", "content": prompt},736        ...             {737        ...                 "role": "user",738        ...                 "content": f"Context: {inputs['context']}"739        ...                 f"\n\ninputs['question']",740        ...             },741        ...         ],742        ...     )743        ...     return {"output": completion.choices[0].message.content}744        >>> results_1 = evaluate(745        ...     functools.partial(predict, prompt=prompt_1),746        ...     data=dataset_name,747        ...     description="Evaluating our basic system prompt.",748        ...     blocking=False,  # Run these experiments in parallel749        ... )  # doctest: +ELLIPSIS750        View the evaluation results for experiment:...751        >>> results_2 = evaluate(752        ...     functools.partial(predict, prompt=prompt_2),753        ...     data=dataset_name,754        ...     description="Evaluating our advanced system prompt.",755        ...     blocking=False,756        ... )  # doctest: +ELLIPSIS757        View the evaluation results for experiment:...758        >>> results_1.wait()759        >>> results_2.wait()760 761            Finally, you would compare the two prompts directly:762        >>> import json763        >>> from langsmith.evaluation import evaluate_comparative764        >>> from langsmith import schemas765        >>> def score_preferences(runs: list, example: schemas.Example):766        ...     assert len(runs) == 2  # Comparing 2 systems767        ...     assert isinstance(example, schemas.Example)768        ...     assert all(run.reference_example_id == example.id for run in runs)769        ...     pred_a = runs[0].outputs["output"] if runs[0].outputs else ""770        ...     pred_b = runs[1].outputs["output"] if runs[1].outputs else ""771        ...     ground_truth = example.outputs["answer"] if example.outputs else ""772        ...     tools = [773        ...         {774        ...             "type": "function",775        ...             "function": {776        ...                 "name": "rank_preferences",777        ...                 "description": "Saves the prefered response ('A' or 'B')",778        ...                 "parameters": {779        ...                     "type": "object",780        ...                     "properties": {781        ...                         "reasoning": {782        ...                             "type": "string",783        ...                             "description": "The reasoning behind the choice.",784        ...                         },785        ...                         "preferred_option": {786        ...                             "type": "string",787        ...                             "enum": ["A", "B"],788        ...                             "description": "The preferred option, either 'A' or 'B'",789        ...                         },790        ...                     },791        ...                     "required": ["preferred_option"],792        ...                 },793        ...             },794        ...         }795        ...     ]796        ...     completion = openai.Client().chat.completions.create(797        ...         model="gpt-4o-mini",798        ...         messages=[799        ...             {"role": "system", "content": "Select the better response."},800        ...             {801        ...                 "role": "user",802        ...                 "content": f"Option A: {pred_a}"803        ...                 f"\n\nOption B: {pred_b}"804        ...                 f"\n\nGround Truth: {ground_truth}",805        ...             },806        ...         ],807        ...         tools=tools,808        ...         tool_choice={809        ...             "type": "function",810        ...             "function": {"name": "rank_preferences"},811        ...         },812        ...     )813        ...     tool_args = completion.choices[0].message.tool_calls[0].function.arguments814        ...     loaded_args = json.loads(tool_args)815        ...     preference = loaded_args["preferred_option"]816        ...     comment = loaded_args["reasoning"]817        ...     if preference == "A":818        ...         return {819        ...             "key": "ranked_preference",820        ...             "scores": {runs[0].id: 1, runs[1].id: 0},821        ...             "comment": comment,822        ...         }823        ...     else:824        ...         return {825        ...             "key": "ranked_preference",826        ...             "scores": {runs[0].id: 0, runs[1].id: 1},827        ...             "comment": comment,828        ...         }829        >>> def score_length_difference(runs: list, example: schemas.Example):830        ...     # Just return whichever response is longer.831        ...     # Just an example, not actually useful in real life.832        ...     assert len(runs) == 2  # Comparing 2 systems833        ...     assert isinstance(example, schemas.Example)834        ...     assert all(run.reference_example_id == example.id for run in runs)835        ...     pred_a = runs[0].outputs["output"] if runs[0].outputs else ""836        ...     pred_b = runs[1].outputs["output"] if runs[1].outputs else ""837        ...     if len(pred_a) > len(pred_b):838        ...         return {839        ...             "key": "length_difference",840        ...             "scores": {runs[0].id: 1, runs[1].id: 0},841        ...         }842        ...     else:843        ...         return {844        ...             "key": "length_difference",845        ...             "scores": {runs[0].id: 0, runs[1].id: 1},846        ...         }847        >>> results = evaluate_comparative(848        ...     [results_1.experiment_name, results_2.experiment_name],849        ...     evaluators=[score_preferences, score_length_difference],850        ...     client=client,851        ... )  # doctest: +ELLIPSIS852        View the pairwise evaluation results at:...853        >>> eval_results = list(results)854        >>> assert len(eval_results) >= 10  # doctest: +SKIP855        >>> assert all(856        ...     "feedback.ranked_preference" in r["evaluation_results"]857        ...     for r in eval_results858        ... )  # doctest: +SKIP859        >>> assert all(860        ...     "feedback.length_difference" in r["evaluation_results"]861        ...     for r in eval_results862        ... )  # doctest: +SKIP863    """  # noqa: E501864    if len(experiments) < 2:865        raise ValueError("Comparative evaluation requires at least 2 experiments.")866    if not evaluators:867        raise ValueError(868            "At least one evaluator is required for comparative evaluation."869        )870    if max_concurrency < 0:871        raise ValueError("max_concurrency must be a positive integer.")872    client = client or rt.get_cached_client()873 874    # TODO: Add information about comparison experiments875    projects = [_load_experiment(experiment, client) for experiment in experiments]876    ref_datasets_ = [str(p.reference_dataset_id) for p in projects]877    if not len(set(ref_datasets_)) == 1:878        raise ValueError("All experiments must have the same reference dataset.")879    experiment_ids = [p.id for p in projects]880    if experiment_prefix is None:881        experiment_names = [p.name for p in projects if p.name is not None]882        experiment_name = (883            " vs. ".join(experiment_names) + "-" + str(uuid.uuid4().hex[:4])884        )885    else:886        experiment_name = experiment_prefix + "-" + str(uuid.uuid4().hex[:8])887    comparative_experiment_id = uuid.uuid4()888    comparative_experiment = client.create_comparative_experiment(889        experiment_name,890        experiments=experiment_ids,891        description=description,892        metadata=metadata,893        id=comparative_experiment_id,894    )895    _print_comparative_experiment_start(896        cast(897            tuple[schemas.TracerSessionResult, schemas.TracerSessionResult],898            tuple(projects),899        ),900        comparative_experiment,901    )902    runs = [903        _load_traces(experiment, client, load_nested=load_nested)904        for experiment in experiments905    ]906    # Only check intersections for the experiments907    examples_intersection = None908    for runs_list in runs:909        example_ids_set = {run.reference_example_id for run in runs_list}910        if examples_intersection is None:911            examples_intersection = example_ids_set912        else:913            examples_intersection &= example_ids_set914    example_ids_nullable = (915        list(examples_intersection) if examples_intersection is not None else []916    )917    example_ids = [eid for eid in example_ids_nullable if eid is not None]918    # TODO: Warn if different dataset versions, etc. are used in the different919    # experiments. We aren't providing any training wheels here.920    batch_size = 99921    data = {}922    for i in range(0, len(example_ids), batch_size):923        example_ids_batch = example_ids[i : i + batch_size]924        for e in client.list_examples(925            dataset_id=projects[0].reference_dataset_id,926            as_of=projects[0].metadata.get("dataset_version"),927            example_ids=example_ids_batch,928        ):929            data[e.id] = e930    runs_dict: dict[uuid.UUID, list[schemas.Run]] = collections.defaultdict(list)931    for runs_list in runs:932        for run in runs_list:933            if run.reference_example_id in data:934                runs_dict[cast(uuid.UUID, run.reference_example_id)].append(run)935 936    comparators = [comparison_evaluator(evaluator) for evaluator in evaluators or []]937    results: dict = {}938 939    def evaluate_and_submit_feedback(940        runs_list: list[schemas.Run],941        example: schemas.Example,942        comparator: DynamicComparisonRunEvaluator,943        executor: cf.Executor,944    ) -> tuple[uuid.UUID, ComparisonEvaluationResult]:945        feedback_group_id = uuid.uuid4()946        if randomize_order:947            random.shuffle(runs_list)948        with rh.tracing_context(project_name="evaluators", client=client):949            result = comparator.compare_runs(runs_list, example)950            if client is None:951                raise ValueError("Client is required to submit feedback.")952        comments = (953            {str(rid): result.comment for rid in result.scores}954            if isinstance(result.comment, str)955            else (result.comment or {})956        )957        # Build a lookup for run metadata958        runs_by_id = {str(run.id): run for run in runs_list}959        for run_id, score in result.scores.items():960            run = runs_by_id.get(str(run_id))961            executor.submit(962                client.create_feedback,963                run_id=run_id,964                key=result.key,965                score=score,966                comment=comments.get(str(run_id)),967                comparative_experiment_id=comparative_experiment.id,968                source_run_id=result.source_run_id,969                feedback_group_id=feedback_group_id,970                session_id=run.session_id if run else None,971                start_time=run.start_time if run else None,972            )973        return example.id, result974 975    tqdm = _load_tqdm()976    with ls_utils.ContextThreadPoolExecutor(977        max_workers=max_concurrency or 1978    ) as executor:979        futures = []980        for example_id, runs_list in tqdm(runs_dict.items()):981            results[example_id] = {"runs": runs_list}982            for comparator in comparators:983                if max_concurrency > 1:984                    future = executor.submit(985                        evaluate_and_submit_feedback,986                        runs_list,987                        data[example_id],988                        comparator,989                        executor,990                    )991                    futures.append(future)992                else:993                    _, result = evaluate_and_submit_feedback(994                        runs_list, data[example_id], comparator, executor995                    )996                    results[example_id][f"feedback.{result.key}"] = result997        if futures:998            cf.wait(futures)999            for future in futures:1000                example_id, result = future.result()1001                results[example_id][f"feedback.{result.key}"] = result1002 1003    return ComparativeExperimentResults(results, data)1004 1005 1006class ComparativeExperimentResults:1007    """Represents the results of an evaluate_comparative() call.1008 1009    This class provides an iterator interface to iterate over the experiment results1010    as they become available. It also provides methods to access the experiment name,1011    the number of results, and to wait for the results to be processed.1012 1013    Methods:1014        experiment_name() -> str: Returns the name of the experiment.1015        wait() -> None: Waits for the experiment data to be processed.1016    """1017 1018    def __init__(1019        self,1020        results: dict,1021        examples: Optional[dict[uuid.UUID, schemas.Example]] = None,1022    ):1023        self._results = results1024        self._examples = examples1025 1026    def __getitem__(self, key):1027        """Return the result associated with the given key."""1028        return self._results[key]1029 1030    def __iter__(self):1031        for key, value in self._results.items():1032            yield {1033                "example": self._examples[key] if self._examples else None,1034                "evaluation_results": value,1035            }1036 1037 1038## Private API1039 1040 1041def _print_comparative_experiment_start(1042    experiments: tuple[schemas.TracerSession, schemas.TracerSession],1043    comparative_experiment: schemas.ComparativeExperiment,1044) -> None:1045    url = experiments[0].url or experiments[1].url1046    if url:1047        project_url = url.split("?")[0]1048        dataset_id = comparative_experiment.reference_dataset_id1049        base_url = project_url.split("/projects/p/")[0]1050        comparison_url = (1051            f"{base_url}/datasets/{dataset_id}/compare?"1052            f"selectedSessions={'%2C'.join([str(e.id) for e in experiments])}"1053            f"&comparativeExperiment={comparative_experiment.id}"1054        )1055        print(  # noqa: T2011056            f"View the pairwise evaluation results at:\n{comparison_url}\n\n"1057        )1058 1059 1060def _is_callable(target: Union[TARGET_T, Iterable[schemas.Run], Runnable]) -> bool:1061    return callable(target) or _is_langchain_runnable(target)1062 1063 1064def _evaluate(1065    target: Union[TARGET_T, Iterable[schemas.Run], Runnable],1066    /,1067    data: DATA_T,1068    evaluators: Optional[Sequence[EVALUATOR_T]] = None,1069    summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,1070    metadata: Optional[dict] = None,1071    experiment_prefix: Optional[str] = None,1072    description: Optional[str] = None,1073    max_concurrency: Optional[int] = None,1074    num_repetitions: int = 1,1075    client: Optional[langsmith.Client] = None,1076    blocking: bool = True,1077    experiment: Optional[Union[schemas.TracerSession, str, uuid.UUID]] = None,1078    upload_results: bool = True,1079    error_handling: Literal["log", "ignore"] = "log",1080) -> ExperimentResults:1081    # Initialize the experiment manager.1082    client = client or rt.get_cached_client()1083    runs = None if _is_callable(target) else cast(Iterable[schemas.Run], target)1084    experiment_, runs = _resolve_experiment(experiment, runs, client)1085 1086    manager = _ExperimentManager(1087        data,1088        client=client,1089        metadata=metadata,1090        experiment=experiment_ or experiment_prefix,1091        description=description,1092        num_repetitions=num_repetitions,1093        # If provided, we don't need to create a new experiment.1094        runs=runs,1095        # Create or resolve the experiment.1096        include_attachments=_include_attachments(target, evaluators),1097        upload_results=upload_results,1098        error_handling=error_handling,1099    ).start()1100    if cache_dir := ls_utils.get_cache_dir(None):1101        cache_path = pathlib.Path(cache_dir) / f"{manager.dataset_id}.yaml"1102    else:1103        cache_path = None1104    with ls_utils.with_optional_cache(cache_path, ignore_hosts=[client.api_url]):1105        if _is_callable(target):1106            # Add predictions to the experiment.1107            manager = manager.with_predictions(1108                cast(TARGET_T, target), max_concurrency=max_concurrency1109            )1110        if evaluators:1111            # Apply evaluators to the predictions.1112            manager = manager.with_evaluators(1113                evaluators, max_concurrency=max_concurrency1114            )1115        if summary_evaluators:1116            # Apply the experiment-level summary evaluators.1117            manager = manager.with_summary_evaluators(summary_evaluators)1118        # Start consuming the results.1119        results = ExperimentResults(manager, blocking=blocking)1120        return results1121 1122 1123def _is_uuid(value: str) -> bool:1124    try:1125        uuid.UUID(value)1126        return True1127    except ValueError:1128        return False1129 1130 1131def _load_experiment(1132    project: EXPERIMENT_T, client: langsmith.Client1133) -> schemas.TracerSession:1134    if isinstance(project, schemas.TracerSession):1135        return project1136    elif isinstance(project, uuid.UUID) or _is_uuid(project):1137        return client.read_project(project_id=project)1138    else:1139        return client.read_project(project_name=project)1140 1141 1142def _load_traces(1143    project: Union[str, uuid.UUID, schemas.TracerSession],1144    client: langsmith.Client,1145    load_nested: bool = False,1146) -> list[schemas.Run]:1147    """Load nested traces for a given project."""1148    is_root = None if load_nested else True1149    if isinstance(project, schemas.TracerSession):1150        runs = client.list_runs(project_id=project.id, is_root=is_root)1151    elif isinstance(project, uuid.UUID) or _is_uuid(project):1152        runs = client.list_runs(project_id=project, is_root=is_root)1153    else:1154        runs = client.list_runs(project_name=project, is_root=is_root)1155    if not load_nested:1156        return list(runs)1157 1158    treemap: collections.defaultdict[uuid.UUID, list[schemas.Run]] = (1159        collections.defaultdict(list)1160    )1161    results = []1162    all_runs = {}1163    for run in runs:1164        if run.parent_run_id is not None:1165            treemap[run.parent_run_id].append(run)1166        else:1167            results.append(run)1168        all_runs[run.id] = run1169    for run_id, child_runs in treemap.items():1170        all_runs[run_id].child_runs = sorted(child_runs, key=lambda r: r.dotted_order)1171    return results1172 1173 1174def _load_examples_map(1175    client: langsmith.Client, project: schemas.TracerSession1176) -> dict[uuid.UUID, schemas.Example]:1177    return {1178        e.id: e1179        for e in client.list_examples(1180            dataset_id=project.reference_dataset_id,1181            as_of=project.metadata.get("dataset_version"),1182        )1183    }1184 1185 1186IT = TypeVar("IT")1187 1188 1189def _load_tqdm() -> Callable[[IT], IT]:1190    try:1191        from tqdm.auto import tqdm1192    except ImportError:1193        return lambda x: x1194    return tqdm  # type: ignore[return-value]1195 1196 1197ET = TypeVar("ET", bound="_ExperimentManagerMixin")1198 1199 1200class _ExperimentManagerMixin:

Showing the first 1,200 of 2315 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai