codekingpro/portable-devtools
114k
1"""V2 Evaluation Interface."""2 3from __future__ import annotations4 5import ast6import collections7import concurrent.futures as cf8import functools9import inspect10import io11import itertools12import logging13import pathlib14import queue15import random16import textwrap17import threading18import uuid19from collections.abc import Awaitable, Generator, Iterable, Iterator, Sequence20from contextvars import copy_context21from typing import (22 TYPE_CHECKING,23 Any,24 Callable,25 Literal,26 Optional,27 TypeVar,28 Union,29 cast,30)31 32from typing_extensions import TypedDict, overload33 34import langsmith35from langsmith import env as ls_env36from langsmith import run_helpers as rh37from langsmith import run_trees as rt38from langsmith import schemas39from langsmith import utils as ls_utils40from langsmith._internal._beta_decorator import _warn_once41from langsmith.evaluation.evaluator import (42 SUMMARY_EVALUATOR_T,43 ComparisonEvaluationResult,44 DynamicComparisonRunEvaluator,45 DynamicRunEvaluator,46 EvaluationResult,47 EvaluationResults,48 RunEvaluator,49 _normalize_summary_evaluator,50 comparison_evaluator,51 run_evaluator,52)53 54# Python 3.14+ removes ast.Str in favor of ast.Constant55_AST_STR_TYPES: tuple = (56 (ast.Str, ast.Constant) if hasattr(ast, "Str") else (ast.Constant,)57)58 59 60def _get_str_value(node: ast.expr) -> str:61 """Get string value from ast.Str or ast.Constant."""62 return node.value if isinstance(node, ast.Constant) else node.s # type: ignore[return-value,union-attr,attr-defined]63 64 65if TYPE_CHECKING:66 import pandas as pd67 from langchain_core.runnables import Runnable68 69 DataFrame = pd.DataFrame70else:71 DataFrame = Any72logger = logging.getLogger(__name__)73 74TARGET_T = Union[Callable[[dict], dict], Callable[[dict, dict], dict]]75# Data format: dataset-name, dataset_id, or examples76DATA_T = Union[str, uuid.UUID, Iterable[schemas.Example], schemas.Dataset]77# Summary evaluator runs over the whole dataset78# and reports aggregate metric(s)79# Row-level evaluator80EVALUATOR_T = Union[81 RunEvaluator,82 Callable[83 [schemas.Run, Optional[schemas.Example]],84 Union[EvaluationResult, EvaluationResults],85 ],86 Callable[..., Union[dict, EvaluationResults, EvaluationResult]],87]88AEVALUATOR_T = Union[89 Callable[90 [schemas.Run, Optional[schemas.Example]],91 Awaitable[Union[EvaluationResult, EvaluationResults]],92 ],93]94EXPERIMENT_T = Union[str, uuid.UUID, schemas.TracerSession]95 96 97@overload98def evaluate(99 target: Union[TARGET_T, Runnable, EXPERIMENT_T],100 /,101 data: Optional[DATA_T] = None,102 evaluators: Optional[Sequence[EVALUATOR_T]] = None,103 summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,104 metadata: Optional[dict] = None,105 experiment_prefix: Optional[str] = None,106 description: Optional[str] = None,107 max_concurrency: Optional[int] = 0,108 num_repetitions: int = 1,109 client: Optional[langsmith.Client] = None,110 blocking: bool = True,111 experiment: Optional[EXPERIMENT_T] = None,112 upload_results: bool = True,113 **kwargs: Any,114) -> ExperimentResults: ...115 116 117@overload118def evaluate(119 target: Union[tuple[EXPERIMENT_T, EXPERIMENT_T]],120 /,121 data: Optional[DATA_T] = None,122 evaluators: Optional[Sequence[COMPARATIVE_EVALUATOR_T]] = None,123 summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,124 metadata: Optional[dict] = None,125 experiment_prefix: Optional[str] = None,126 description: Optional[str] = None,127 max_concurrency: Optional[int] = 0,128 num_repetitions: int = 1,129 client: Optional[langsmith.Client] = None,130 blocking: bool = True,131 experiment: Optional[EXPERIMENT_T] = None,132 upload_results: bool = True,133 **kwargs: Any,134) -> ComparativeExperimentResults: ...135 136 137def evaluate(138 target: Union[TARGET_T, Runnable, EXPERIMENT_T, tuple[EXPERIMENT_T, EXPERIMENT_T]],139 /,140 data: Optional[DATA_T] = None,141 evaluators: Optional[142 Union[Sequence[EVALUATOR_T], Sequence[COMPARATIVE_EVALUATOR_T]]143 ] = None,144 summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,145 metadata: Optional[dict] = None,146 experiment_prefix: Optional[str] = None,147 description: Optional[str] = None,148 max_concurrency: Optional[int] = 0,149 num_repetitions: int = 1,150 client: Optional[langsmith.Client] = None,151 blocking: bool = True,152 experiment: Optional[EXPERIMENT_T] = None,153 upload_results: bool = True,154 error_handling: Literal["log", "ignore"] = "log",155 **kwargs: Any,156) -> Union[ExperimentResults, ComparativeExperimentResults]:157 r"""Evaluate a target system on a given dataset.158 159 Args:160 target (TARGET_T | Runnable | EXPERIMENT_T | Tuple[EXPERIMENT_T, EXPERIMENT_T]):161 The target system or experiment(s) to evaluate.162 163 Can be a function that takes a dict and returns a `dict`, a langchain `Runnable`, an164 existing experiment ID, or a two-tuple of experiment IDs.165 data (DATA_T): The dataset to evaluate on.166 167 Can be a dataset name, a list of examples, or a generator of examples.168 evaluators (Sequence[EVALUATOR_T] | Sequence[COMPARATIVE_EVALUATOR_T] | None):169 A list of evaluators to run on each example. The evaluator signature170 depends on the target type.171 summary_evaluators (Sequence[SUMMARY_EVALUATOR_T] | None): A list of summary172 evaluators to run on the entire dataset.173 174 Should not be specified if comparing two existing experiments.175 metadata (dict | None): Metadata to attach to the experiment.176 experiment_prefix (str | None): A prefix to provide for your experiment name.177 description (str | None): A free-form text description for the experiment.178 max_concurrency (int | None): The maximum number of concurrent179 evaluations to run.180 181 If `None` then no limit is set. If `0` then no concurrency.182 client (langsmith.Client | None): The LangSmith client to use.183 blocking (bool): Whether to block until the evaluation is complete.184 num_repetitions (int): The number of times to run the evaluation.185 Each item in the dataset will be run and evaluated this many times.186 experiment (schemas.TracerSession | None): An existing experiment to187 extend.188 189 If provided, `experiment_prefix` is ignored.190 191 For advanced usage only. Should not be specified if target is an existing192 experiment or two-tuple fo experiments.193 error_handling (str, default="log"): How to handle individual run errors.194 195 `'log'` will trace the runs with the error message as part of the196 experiment, `'ignore'` will not count the run as part of the experiment at197 all.198 199 Returns:200 ExperimentResults: If target is a function, `Runnable`, or existing experiment.201 ComparativeExperimentResults: If target is a two-tuple of existing experiments.202 203 Examples:204 Prepare the dataset:205 206 >>> from typing import Sequence207 >>> from langsmith import Client208 >>> from langsmith.evaluation import evaluate209 >>> from langsmith.schemas import Example, Run210 >>> client = Client()211 >>> dataset = client.clone_public_dataset(212 ... "https://smith.langchain.com/public/419dcab2-1d66-4b94-8901-0357ead390df/d"213 ... )214 >>> dataset_name = "Evaluate Examples"215 216 Basic usage:217 218 >>> def accuracy(run: Run, example: Example):219 ... # Row-level evaluator for accuracy.220 ... pred = run.outputs["output"]221 ... expected = example.outputs["answer"]222 ... return {"score": expected.lower() == pred.lower()}223 >>> def precision(runs: Sequence[Run], examples: Sequence[Example]):224 ... # Experiment-level evaluator for precision.225 ... # TP / (TP + FP)226 ... predictions = [run.outputs["output"].lower() for run in runs]227 ... expected = [example.outputs["answer"].lower() for example in examples]228 ... # yes and no are the only possible answers229 ... tp = sum([p == e for p, e in zip(predictions, expected) if p == "yes"])230 ... fp = sum([p == "yes" and e == "no" for p, e in zip(predictions, expected)])231 ... return {"score": tp / (tp + fp)}232 >>> def predict(inputs: dict) -> dict:233 ... # This can be any function or just an API call to your app.234 ... return {"output": "Yes"}235 >>> results = evaluate(236 ... predict,237 ... data=dataset_name,238 ... evaluators=[accuracy],239 ... summary_evaluators=[precision],240 ... experiment_prefix="My Experiment",241 ... description="Evaluating the accuracy of a simple prediction model.",242 ... metadata={243 ... "my-prompt-version": "abcd-1234",244 ... },245 ... ) # doctest: +ELLIPSIS246 View the evaluation results for experiment:...247 248 Evaluating over only a subset of the examples249 250 >>> experiment_name = results.experiment_name251 >>> examples = client.list_examples(dataset_name=dataset_name, limit=5)252 >>> results = evaluate(253 ... predict,254 ... data=examples,255 ... evaluators=[accuracy],256 ... summary_evaluators=[precision],257 ... experiment_prefix="My Experiment",258 ... description="Just testing a subset synchronously.",259 ... ) # doctest: +ELLIPSIS260 View the evaluation results for experiment:...261 262 Streaming each prediction to more easily + eagerly debug.263 264 >>> results = evaluate(265 ... predict,266 ... data=dataset_name,267 ... evaluators=[accuracy],268 ... summary_evaluators=[precision],269 ... description="I don't even have to block!",270 ... blocking=False,271 ... ) # doctest: +ELLIPSIS272 View the evaluation results for experiment:...273 >>> for i, result in enumerate(results): # doctest: +ELLIPSIS274 ... pass275 276 277 278 Evaluating a LangChain object:279 280 >>> from langchain_core.runnables import chain as as_runnable281 >>> @as_runnable282 ... def nested_predict(inputs):283 ... return {"output": "Yes"}284 >>> @as_runnable285 ... def lc_predict(inputs):286 ... return nested_predict.invoke(inputs)287 >>> results = evaluate(288 ... lc_predict.invoke,289 ... data=dataset_name,290 ... evaluators=[accuracy],291 ... description="This time we're evaluating a LangChain object.",292 ... summary_evaluators=[precision],293 ... ) # doctest: +ELLIPSIS294 View the evaluation results for experiment:...295 296 !!! warning "Behavior changed in `langsmith` 0.2.0"297 298 'max_concurrency' default updated from None (no limit on concurrency)299 to 0 (no concurrency at all).300 """ # noqa: E501301 if isinstance(target, (str, uuid.UUID, schemas.TracerSession)):302 invalid_args = {303 "num_repetitions": num_repetitions > 1,304 "experiment": bool(experiment),305 "upload_results": not upload_results,306 "experiment_prefix": bool(experiment_prefix),307 "data": bool(data),308 }309 if any(invalid_args.values()):310 msg = (311 f"Received invalid arguments. "312 f"{tuple(k for k, v in invalid_args.items() if v)} should not be "313 f"specified when target is an existing experiment."314 )315 raise ValueError(msg)316 target_id = target if isinstance(target, (str, uuid.UUID)) else target.id317 logger.debug(f"Running evaluation over existing experiment {target_id}...")318 return evaluate_existing(319 target,320 evaluators=cast(Optional[Sequence[EVALUATOR_T]], evaluators),321 summary_evaluators=summary_evaluators,322 metadata=metadata,323 max_concurrency=max_concurrency,324 client=client,325 blocking=blocking,326 **kwargs,327 )328 elif isinstance(target, (list, tuple)):329 invalid_args = {330 "num_repetitions": num_repetitions > 1,331 "experiment": bool(experiment),332 "upload_results": not upload_results,333 "summary_evaluators": bool(summary_evaluators),334 "data": bool(data),335 }336 if len(target) != 2 or not all(337 isinstance(t, (str, uuid.UUID, schemas.TracerSession)) for t in target338 ):339 msg = (340 "Received invalid target. If a tuple is specified it must have length "341 "2 and each element should by the ID or schemas.TracerSession of an "342 f"existing experiment. Received {target=}"343 )344 raise ValueError(msg)345 elif any(invalid_args.values()):346 msg = (347 f"Received invalid arguments. "348 f"{tuple(k for k, v in invalid_args.items() if v)} should not be "349 f"specified when target is two existing experiments."350 )351 raise ValueError(msg)352 if max_concurrency is not None:353 kwargs["max_concurrency"] = max_concurrency354 target_ids = [t if isinstance(t, (str, uuid.UUID)) else t.id for t in target]355 logger.debug(356 f"Running pairwise evaluation over existing experiments {target_ids}..."357 )358 return evaluate_comparative(359 target,360 evaluators=cast(Sequence[COMPARATIVE_EVALUATOR_T], evaluators or ()),361 experiment_prefix=experiment_prefix,362 description=description,363 client=client,364 metadata=metadata,365 **kwargs,366 )367 elif kwargs:368 msg = (369 f"Received unsupported arguments {kwargs}. These arguments are not "370 f"supported when creating a new experiment."371 )372 raise ValueError(msg)373 elif not data:374 msg = "Must specify 'data' when running evaluations over a target function."375 raise ValueError(msg)376 elif callable(target) and rh.is_async(target):377 msg = (378 "Async functions are not supported by `evaluate`. "379 "Please use `aevaluate` instead:\n\n"380 "from langsmith import aevaluate\n\n"381 "await aevaluate(\n"382 " async_target_function,\n"383 " data=data,\n"384 " evaluators=evaluators,\n"385 " # ... other parameters\n"386 ")"387 )388 raise ValueError(msg)389 elif experiment and experiment_prefix:390 msg = (391 "Expected at most one of 'experiment' or 'experiment_prefix',"392 " but both were provided. "393 f"Got: experiment={experiment}, experiment_prefix={experiment_prefix}"394 )395 raise ValueError(msg)396 else:397 if not upload_results:398 _warn_once("'upload_results' parameter is in beta.")399 logger.debug(f"Running evaluation over target system {target}...")400 return _evaluate(401 target,402 data=data,403 evaluators=cast(Optional[Sequence[EVALUATOR_T]], evaluators),404 summary_evaluators=summary_evaluators,405 metadata=metadata,406 experiment_prefix=experiment_prefix,407 description=description,408 max_concurrency=max_concurrency,409 num_repetitions=num_repetitions,410 client=client,411 blocking=blocking,412 experiment=experiment,413 upload_results=upload_results,414 error_handling=error_handling,415 )416 417 418def evaluate_existing(419 experiment: Union[str, uuid.UUID, schemas.TracerSession],420 /,421 evaluators: Optional[Sequence[EVALUATOR_T]] = None,422 summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,423 metadata: Optional[dict] = None,424 max_concurrency: Optional[int] = 0,425 client: Optional[langsmith.Client] = None,426 load_nested: bool = False,427 blocking: bool = True,428) -> ExperimentResults:429 r"""Evaluate existing experiment runs.430 431 Args:432 experiment (Union[str, uuid.UUID]): The identifier of the experiment to evaluate.433 evaluators (Optional[Sequence[EVALUATOR_T]]): Optional sequence of evaluators to use for individual run evaluation.434 summary_evaluators (Optional[Sequence[SUMMARY_EVALUATOR_T]]): Optional sequence of evaluators435 to apply over the entire dataset.436 metadata (Optional[dict]): Optional metadata to include in the evaluation results.437 max_concurrency (int | None): The maximum number of concurrent438 evaluations to run.439 440 If `None` then no limit is set. If `0` then no concurrency.441 client (Optional[langsmith.Client]): Optional Langsmith client to use for evaluation.442 load_nested: Whether to load all child runs for the experiment.443 444 Default is to only load the top-level root runs.445 blocking (bool): Whether to block until evaluation is complete.446 447 Returns:448 The evaluation results.449 450 Environment:451 - `LANGSMITH_TEST_CACHE`: If set, API calls will be cached to disk to save time and452 cost during testing.453 454 Recommended to commit the cache files to your repository for faster CI/CD runs.455 456 Requires the `'langsmith[vcr]'` package to be installed.457 458 Examples:459 Define your evaluators460 461 >>> from typing import Sequence462 >>> from langsmith.schemas import Example, Run463 >>> def accuracy(run: Run, example: Example):464 ... # Row-level evaluator for accuracy.465 ... pred = run.outputs["output"]466 ... expected = example.outputs["answer"]467 ... return {"score": expected.lower() == pred.lower()}468 >>> def precision(runs: Sequence[Run], examples: Sequence[Example]):469 ... # Experiment-level evaluator for precision.470 ... # TP / (TP + FP)471 ... predictions = [run.outputs["output"].lower() for run in runs]472 ... expected = [example.outputs["answer"].lower() for example in examples]473 ... # yes and no are the only possible answers474 ... tp = sum([p == e for p, e in zip(predictions, expected) if p == "yes"])475 ... fp = sum([p == "yes" and e == "no" for p, e in zip(predictions, expected)])476 ... return {"score": tp / (tp + fp)}477 478 Load the experiment and run the evaluation.479 480 >>> import uuid481 >>> from langsmith import Client482 >>> from langsmith.evaluation import evaluate, evaluate_existing483 >>> client = Client()484 >>> dataset_name = "__doctest_evaluate_existing_" + uuid.uuid4().hex[:8]485 >>> dataset = client.create_dataset(dataset_name)486 >>> example = client.create_example(487 ... inputs={"question": "What is 2+2?"},488 ... outputs={"answer": "4"},489 ... dataset_id=dataset.id,490 ... )491 >>> def predict(inputs: dict) -> dict:492 ... return {"output": "4"}493 >>> # First run inference on the dataset494 ... results = evaluate(495 ... predict, data=dataset_name, experiment_prefix="doctest_experiment"496 ... ) # doctest: +ELLIPSIS497 View the evaluation results for experiment:...498 >>> experiment_id = results.experiment_name499 >>> # Wait for the experiment to be fully processed and check if we have results500 >>> len(results) > 0501 True502 >>> import time503 >>> time.sleep(5) # Wait longer for runs to be indexed504 >>> results = evaluate_existing(505 ... experiment_id,506 ... evaluators=[accuracy],507 ... summary_evaluators=[precision],508 ... ) # doctest: +ELLIPSIS509 View the evaluation results for experiment:...510 >>> client.delete_dataset(dataset_id=dataset.id)511 """ # noqa: E501512 client = client or rt.get_cached_client(timeout_ms=(20_000, 90_001))513 project = _load_experiment(experiment, client)514 runs = _load_traces(experiment, client, load_nested=load_nested)515 data_map = _load_examples_map(client, project)516 data = [data_map[cast(uuid.UUID, run.reference_example_id)] for run in runs]517 return _evaluate(518 runs,519 data=data,520 evaluators=evaluators,521 summary_evaluators=summary_evaluators,522 metadata=metadata,523 max_concurrency=max_concurrency,524 client=client,525 blocking=blocking,526 experiment=project,527 )528 529 530class ExperimentResultRow(TypedDict):531 run: schemas.Run532 example: schemas.Example533 evaluation_results: EvaluationResults534 535 536class ExperimentResults:537 """Represents the results of an evaluate() call.538 539 This class provides an iterator interface to iterate over the experiment results540 as they become available. It also provides methods to access the experiment name,541 the number of results, and to wait for the results to be processed.542 543 Methods:544 experiment_name() -> str: Returns the name of the experiment.545 wait() -> None: Waits for the experiment data to be processed.546 """547 548 def __init__(self, experiment_manager: _ExperimentManager, blocking: bool = True):549 self._manager = experiment_manager550 self._results: list[ExperimentResultRow] = []551 self._queue: queue.Queue[ExperimentResultRow] = queue.Queue()552 self._processing_complete = threading.Event()553 self._processing_error: Optional[BaseException] = None554 if not blocking:555 self._thread: Optional[threading.Thread] = threading.Thread(556 target=self._process_data557 )558 self._thread.start()559 else:560 self._thread = None561 self._process_data()562 563 @property564 def experiment_name(self) -> str:565 return self._manager.experiment_name566 567 @property568 def experiment_id(self) -> uuid.UUID:569 """The ID of the experiment."""570 return self._manager._get_experiment().id571 572 @property573 def url(self) -> Optional[str]:574 """The URL of the experiment in the LangSmith UI."""575 experiment = self._manager._get_experiment()576 if experiment.url:577 project_url = experiment.url.split("?")[0]578 base_url = project_url.split("/projects/p/")[0]579 return (580 f"{base_url}/datasets/{self._manager.dataset_id}/compare?"581 f"selectedSessions={experiment.id}"582 )583 return None584 585 def get_dataset_id(self) -> str:586 """Get the ID of the dataset associated with this experiment."""587 return self._manager.dataset_id588 589 @property590 def comparison_url(self) -> Optional[str]:591 """The URL to the comparison view for this experiment."""592 return self.url593 594 def __iter__(self) -> Iterator[ExperimentResultRow]:595 ix = 0596 while (597 not self._processing_complete.is_set()598 or not self._queue.empty()599 or ix < len(self._results)600 ):601 try:602 if ix < len(self._results):603 yield self._results[ix]604 ix += 1605 else:606 self._queue.get(block=True, timeout=0.1)607 except queue.Empty:608 if self._processing_error is not None:609 raise self._processing_error610 continue611 if self._processing_error is not None:612 raise self._processing_error613 614 def _process_data(self) -> None:615 tqdm = _load_tqdm()616 try:617 results = self._manager.get_results()618 for item in tqdm(results):619 self._queue.put(item)620 self._results.append(item)621 622 summary_scores = self._manager.get_summary_scores()623 self._summary_results = summary_scores624 except BaseException as e:625 self._processing_error = e626 finally:627 self._processing_complete.set()628 629 def __len__(self) -> int:630 return len(self._results)631 632 def to_pandas(633 self, start: Optional[int] = 0, end: Optional[int] = None634 ) -> DataFrame:635 return _to_pandas(self._results, start=start, end=end)636 637 def _repr_html_(self) -> str:638 import importlib.util639 640 if self._results and importlib.util.find_spec("pandas"):641 df = self.to_pandas()642 return df._repr_html_() # type: ignore[operator]643 else:644 return self.__repr__()645 646 def __repr__(self) -> str:647 return f"<ExperimentResults {self.experiment_name}>"648 649 def wait(self) -> None:650 """Wait for the evaluation runner to complete.651 652 This method blocks the current thread until the evaluation runner has653 finished its execution.654 """655 if self._thread:656 self._thread.join()657 if self._processing_error is not None:658 raise self._processing_error659 660 661## Public API for Comparison Experiments662 663# Row-level evaluator664COMPARATIVE_EVALUATOR_T = Callable[665 [Sequence[schemas.Run], Optional[schemas.Example]],666 Union[667 Union[ComparisonEvaluationResult, dict],668 Awaitable[Union[ComparisonEvaluationResult, dict]],669 ],670]671 672 673def evaluate_comparative(674 experiments: tuple[EXPERIMENT_T, EXPERIMENT_T],675 /,676 evaluators: Sequence[COMPARATIVE_EVALUATOR_T],677 experiment_prefix: Optional[str] = None,678 description: Optional[str] = None,679 max_concurrency: int = 5,680 client: Optional[langsmith.Client] = None,681 metadata: Optional[dict] = None,682 load_nested: bool = False,683 randomize_order: bool = False,684) -> ComparativeExperimentResults:685 r"""Evaluate existing experiment runs against each other.686 687 This lets you use pairwise preference scoring to generate more688 reliable feedback in your experiments.689 690 Args:691 experiments (Tuple[Union[str, uuid.UUID], Union[str, uuid.UUID]]):692 The identifiers of the experiments to compare.693 evaluators (Sequence[COMPARATIVE_EVALUATOR_T]):694 A list of evaluators to run on each example.695 experiment_prefix (Optional[str]): A prefix to provide for your experiment name.696 description (Optional[str]): A free-form text description for the experiment.697 max_concurrency (int): The maximum number of concurrent evaluations to run.698 client (Optional[langsmith.Client]): The LangSmith client to use.699 metadata (Optional[dict]): Metadata to attach to the experiment.700 load_nested (bool): Whether to load all child runs for the experiment.701 702 Default is to only load the top-level root runs.703 randomize_order (bool): Whether to randomize the order of the outputs for each evaluation.704 705 Returns:706 The results of the comparative evaluation.707 708 Examples:709 Suppose you want to compare two prompts to see which one is more effective.710 You would first prepare your dataset:711 712 >>> from typing import Sequence713 >>> from langsmith import Client714 >>> from langsmith.evaluation import evaluate715 >>> from langsmith.schemas import Example, Run716 >>> client = Client()717 >>> dataset = client.clone_public_dataset(718 ... "https://smith.langchain.com/public/419dcab2-1d66-4b94-8901-0357ead390df/d"719 ... )720 >>> dataset_name = "Evaluate Examples"721 722 Then you would run your different prompts:723 >>> import functools724 >>> import openai725 >>> from langsmith.evaluation import evaluate726 >>> from langsmith.wrappers import wrap_openai727 >>> oai_client = openai.Client()728 >>> wrapped_client = wrap_openai(oai_client)729 >>> prompt_1 = "You are a helpful assistant."730 >>> prompt_2 = "You are an exceedingly helpful assistant."731 >>> def predict(inputs: dict, prompt: str) -> dict:732 ... completion = wrapped_client.chat.completions.create(733 ... model="gpt-4o-mini",734 ... messages=[735 ... {"role": "system", "content": prompt},736 ... {737 ... "role": "user",738 ... "content": f"Context: {inputs['context']}"739 ... f"\n\ninputs['question']",740 ... },741 ... ],742 ... )743 ... return {"output": completion.choices[0].message.content}744 >>> results_1 = evaluate(745 ... functools.partial(predict, prompt=prompt_1),746 ... data=dataset_name,747 ... description="Evaluating our basic system prompt.",748 ... blocking=False, # Run these experiments in parallel749 ... ) # doctest: +ELLIPSIS750 View the evaluation results for experiment:...751 >>> results_2 = evaluate(752 ... functools.partial(predict, prompt=prompt_2),753 ... data=dataset_name,754 ... description="Evaluating our advanced system prompt.",755 ... blocking=False,756 ... ) # doctest: +ELLIPSIS757 View the evaluation results for experiment:...758 >>> results_1.wait()759 >>> results_2.wait()760 761 Finally, you would compare the two prompts directly:762 >>> import json763 >>> from langsmith.evaluation import evaluate_comparative764 >>> from langsmith import schemas765 >>> def score_preferences(runs: list, example: schemas.Example):766 ... assert len(runs) == 2 # Comparing 2 systems767 ... assert isinstance(example, schemas.Example)768 ... assert all(run.reference_example_id == example.id for run in runs)769 ... pred_a = runs[0].outputs["output"] if runs[0].outputs else ""770 ... pred_b = runs[1].outputs["output"] if runs[1].outputs else ""771 ... ground_truth = example.outputs["answer"] if example.outputs else ""772 ... tools = [773 ... {774 ... "type": "function",775 ... "function": {776 ... "name": "rank_preferences",777 ... "description": "Saves the prefered response ('A' or 'B')",778 ... "parameters": {779 ... "type": "object",780 ... "properties": {781 ... "reasoning": {782 ... "type": "string",783 ... "description": "The reasoning behind the choice.",784 ... },785 ... "preferred_option": {786 ... "type": "string",787 ... "enum": ["A", "B"],788 ... "description": "The preferred option, either 'A' or 'B'",789 ... },790 ... },791 ... "required": ["preferred_option"],792 ... },793 ... },794 ... }795 ... ]796 ... completion = openai.Client().chat.completions.create(797 ... model="gpt-4o-mini",798 ... messages=[799 ... {"role": "system", "content": "Select the better response."},800 ... {801 ... "role": "user",802 ... "content": f"Option A: {pred_a}"803 ... f"\n\nOption B: {pred_b}"804 ... f"\n\nGround Truth: {ground_truth}",805 ... },806 ... ],807 ... tools=tools,808 ... tool_choice={809 ... "type": "function",810 ... "function": {"name": "rank_preferences"},811 ... },812 ... )813 ... tool_args = completion.choices[0].message.tool_calls[0].function.arguments814 ... loaded_args = json.loads(tool_args)815 ... preference = loaded_args["preferred_option"]816 ... comment = loaded_args["reasoning"]817 ... if preference == "A":818 ... return {819 ... "key": "ranked_preference",820 ... "scores": {runs[0].id: 1, runs[1].id: 0},821 ... "comment": comment,822 ... }823 ... else:824 ... return {825 ... "key": "ranked_preference",826 ... "scores": {runs[0].id: 0, runs[1].id: 1},827 ... "comment": comment,828 ... }829 >>> def score_length_difference(runs: list, example: schemas.Example):830 ... # Just return whichever response is longer.831 ... # Just an example, not actually useful in real life.832 ... assert len(runs) == 2 # Comparing 2 systems833 ... assert isinstance(example, schemas.Example)834 ... assert all(run.reference_example_id == example.id for run in runs)835 ... pred_a = runs[0].outputs["output"] if runs[0].outputs else ""836 ... pred_b = runs[1].outputs["output"] if runs[1].outputs else ""837 ... if len(pred_a) > len(pred_b):838 ... return {839 ... "key": "length_difference",840 ... "scores": {runs[0].id: 1, runs[1].id: 0},841 ... }842 ... else:843 ... return {844 ... "key": "length_difference",845 ... "scores": {runs[0].id: 0, runs[1].id: 1},846 ... }847 >>> results = evaluate_comparative(848 ... [results_1.experiment_name, results_2.experiment_name],849 ... evaluators=[score_preferences, score_length_difference],850 ... client=client,851 ... ) # doctest: +ELLIPSIS852 View the pairwise evaluation results at:...853 >>> eval_results = list(results)854 >>> assert len(eval_results) >= 10 # doctest: +SKIP855 >>> assert all(856 ... "feedback.ranked_preference" in r["evaluation_results"]857 ... for r in eval_results858 ... ) # doctest: +SKIP859 >>> assert all(860 ... "feedback.length_difference" in r["evaluation_results"]861 ... for r in eval_results862 ... ) # doctest: +SKIP863 """ # noqa: E501864 if len(experiments) < 2:865 raise ValueError("Comparative evaluation requires at least 2 experiments.")866 if not evaluators:867 raise ValueError(868 "At least one evaluator is required for comparative evaluation."869 )870 if max_concurrency < 0:871 raise ValueError("max_concurrency must be a positive integer.")872 client = client or rt.get_cached_client()873 874 # TODO: Add information about comparison experiments875 projects = [_load_experiment(experiment, client) for experiment in experiments]876 ref_datasets_ = [str(p.reference_dataset_id) for p in projects]877 if not len(set(ref_datasets_)) == 1:878 raise ValueError("All experiments must have the same reference dataset.")879 experiment_ids = [p.id for p in projects]880 if experiment_prefix is None:881 experiment_names = [p.name for p in projects if p.name is not None]882 experiment_name = (883 " vs. ".join(experiment_names) + "-" + str(uuid.uuid4().hex[:4])884 )885 else:886 experiment_name = experiment_prefix + "-" + str(uuid.uuid4().hex[:8])887 comparative_experiment_id = uuid.uuid4()888 comparative_experiment = client.create_comparative_experiment(889 experiment_name,890 experiments=experiment_ids,891 description=description,892 metadata=metadata,893 id=comparative_experiment_id,894 )895 _print_comparative_experiment_start(896 cast(897 tuple[schemas.TracerSessionResult, schemas.TracerSessionResult],898 tuple(projects),899 ),900 comparative_experiment,901 )902 runs = [903 _load_traces(experiment, client, load_nested=load_nested)904 for experiment in experiments905 ]906 # Only check intersections for the experiments907 examples_intersection = None908 for runs_list in runs:909 example_ids_set = {run.reference_example_id for run in runs_list}910 if examples_intersection is None:911 examples_intersection = example_ids_set912 else:913 examples_intersection &= example_ids_set914 example_ids_nullable = (915 list(examples_intersection) if examples_intersection is not None else []916 )917 example_ids = [eid for eid in example_ids_nullable if eid is not None]918 # TODO: Warn if different dataset versions, etc. are used in the different919 # experiments. We aren't providing any training wheels here.920 batch_size = 99921 data = {}922 for i in range(0, len(example_ids), batch_size):923 example_ids_batch = example_ids[i : i + batch_size]924 for e in client.list_examples(925 dataset_id=projects[0].reference_dataset_id,926 as_of=projects[0].metadata.get("dataset_version"),927 example_ids=example_ids_batch,928 ):929 data[e.id] = e930 runs_dict: dict[uuid.UUID, list[schemas.Run]] = collections.defaultdict(list)931 for runs_list in runs:932 for run in runs_list:933 if run.reference_example_id in data:934 runs_dict[cast(uuid.UUID, run.reference_example_id)].append(run)935 936 comparators = [comparison_evaluator(evaluator) for evaluator in evaluators or []]937 results: dict = {}938 939 def evaluate_and_submit_feedback(940 runs_list: list[schemas.Run],941 example: schemas.Example,942 comparator: DynamicComparisonRunEvaluator,943 executor: cf.Executor,944 ) -> tuple[uuid.UUID, ComparisonEvaluationResult]:945 feedback_group_id = uuid.uuid4()946 if randomize_order:947 random.shuffle(runs_list)948 with rh.tracing_context(project_name="evaluators", client=client):949 result = comparator.compare_runs(runs_list, example)950 if client is None:951 raise ValueError("Client is required to submit feedback.")952 comments = (953 {str(rid): result.comment for rid in result.scores}954 if isinstance(result.comment, str)955 else (result.comment or {})956 )957 # Build a lookup for run metadata958 runs_by_id = {str(run.id): run for run in runs_list}959 for run_id, score in result.scores.items():960 run = runs_by_id.get(str(run_id))961 executor.submit(962 client.create_feedback,963 run_id=run_id,964 key=result.key,965 score=score,966 comment=comments.get(str(run_id)),967 comparative_experiment_id=comparative_experiment.id,968 source_run_id=result.source_run_id,969 feedback_group_id=feedback_group_id,970 session_id=run.session_id if run else None,971 start_time=run.start_time if run else None,972 )973 return example.id, result974 975 tqdm = _load_tqdm()976 with ls_utils.ContextThreadPoolExecutor(977 max_workers=max_concurrency or 1978 ) as executor:979 futures = []980 for example_id, runs_list in tqdm(runs_dict.items()):981 results[example_id] = {"runs": runs_list}982 for comparator in comparators:983 if max_concurrency > 1:984 future = executor.submit(985 evaluate_and_submit_feedback,986 runs_list,987 data[example_id],988 comparator,989 executor,990 )991 futures.append(future)992 else:993 _, result = evaluate_and_submit_feedback(994 runs_list, data[example_id], comparator, executor995 )996 results[example_id][f"feedback.{result.key}"] = result997 if futures:998 cf.wait(futures)999 for future in futures:1000 example_id, result = future.result()1001 results[example_id][f"feedback.{result.key}"] = result1002 1003 return ComparativeExperimentResults(results, data)1004 1005 1006class ComparativeExperimentResults:1007 """Represents the results of an evaluate_comparative() call.1008 1009 This class provides an iterator interface to iterate over the experiment results1010 as they become available. It also provides methods to access the experiment name,1011 the number of results, and to wait for the results to be processed.1012 1013 Methods:1014 experiment_name() -> str: Returns the name of the experiment.1015 wait() -> None: Waits for the experiment data to be processed.1016 """1017 1018 def __init__(1019 self,1020 results: dict,1021 examples: Optional[dict[uuid.UUID, schemas.Example]] = None,1022 ):1023 self._results = results1024 self._examples = examples1025 1026 def __getitem__(self, key):1027 """Return the result associated with the given key."""1028 return self._results[key]1029 1030 def __iter__(self):1031 for key, value in self._results.items():1032 yield {1033 "example": self._examples[key] if self._examples else None,1034 "evaluation_results": value,1035 }1036 1037 1038## Private API1039 1040 1041def _print_comparative_experiment_start(1042 experiments: tuple[schemas.TracerSession, schemas.TracerSession],1043 comparative_experiment: schemas.ComparativeExperiment,1044) -> None:1045 url = experiments[0].url or experiments[1].url1046 if url:1047 project_url = url.split("?")[0]1048 dataset_id = comparative_experiment.reference_dataset_id1049 base_url = project_url.split("/projects/p/")[0]1050 comparison_url = (1051 f"{base_url}/datasets/{dataset_id}/compare?"1052 f"selectedSessions={'%2C'.join([str(e.id) for e in experiments])}"1053 f"&comparativeExperiment={comparative_experiment.id}"1054 )1055 print( # noqa: T2011056 f"View the pairwise evaluation results at:\n{comparison_url}\n\n"1057 )1058 1059 1060def _is_callable(target: Union[TARGET_T, Iterable[schemas.Run], Runnable]) -> bool:1061 return callable(target) or _is_langchain_runnable(target)1062 1063 1064def _evaluate(1065 target: Union[TARGET_T, Iterable[schemas.Run], Runnable],1066 /,1067 data: DATA_T,1068 evaluators: Optional[Sequence[EVALUATOR_T]] = None,1069 summary_evaluators: Optional[Sequence[SUMMARY_EVALUATOR_T]] = None,1070 metadata: Optional[dict] = None,1071 experiment_prefix: Optional[str] = None,1072 description: Optional[str] = None,1073 max_concurrency: Optional[int] = None,1074 num_repetitions: int = 1,1075 client: Optional[langsmith.Client] = None,1076 blocking: bool = True,1077 experiment: Optional[Union[schemas.TracerSession, str, uuid.UUID]] = None,1078 upload_results: bool = True,1079 error_handling: Literal["log", "ignore"] = "log",1080) -> ExperimentResults:1081 # Initialize the experiment manager.1082 client = client or rt.get_cached_client()1083 runs = None if _is_callable(target) else cast(Iterable[schemas.Run], target)1084 experiment_, runs = _resolve_experiment(experiment, runs, client)1085 1086 manager = _ExperimentManager(1087 data,1088 client=client,1089 metadata=metadata,1090 experiment=experiment_ or experiment_prefix,1091 description=description,1092 num_repetitions=num_repetitions,1093 # If provided, we don't need to create a new experiment.1094 runs=runs,1095 # Create or resolve the experiment.1096 include_attachments=_include_attachments(target, evaluators),1097 upload_results=upload_results,1098 error_handling=error_handling,1099 ).start()1100 if cache_dir := ls_utils.get_cache_dir(None):1101 cache_path = pathlib.Path(cache_dir) / f"{manager.dataset_id}.yaml"1102 else:1103 cache_path = None1104 with ls_utils.with_optional_cache(cache_path, ignore_hosts=[client.api_url]):1105 if _is_callable(target):1106 # Add predictions to the experiment.1107 manager = manager.with_predictions(1108 cast(TARGET_T, target), max_concurrency=max_concurrency1109 )1110 if evaluators:1111 # Apply evaluators to the predictions.1112 manager = manager.with_evaluators(1113 evaluators, max_concurrency=max_concurrency1114 )1115 if summary_evaluators:1116 # Apply the experiment-level summary evaluators.1117 manager = manager.with_summary_evaluators(summary_evaluators)1118 # Start consuming the results.1119 results = ExperimentResults(manager, blocking=blocking)1120 return results1121 1122 1123def _is_uuid(value: str) -> bool:1124 try:1125 uuid.UUID(value)1126 return True1127 except ValueError:1128 return False1129 1130 1131def _load_experiment(1132 project: EXPERIMENT_T, client: langsmith.Client1133) -> schemas.TracerSession:1134 if isinstance(project, schemas.TracerSession):1135 return project1136 elif isinstance(project, uuid.UUID) or _is_uuid(project):1137 return client.read_project(project_id=project)1138 else:1139 return client.read_project(project_name=project)1140 1141 1142def _load_traces(1143 project: Union[str, uuid.UUID, schemas.TracerSession],1144 client: langsmith.Client,1145 load_nested: bool = False,1146) -> list[schemas.Run]:1147 """Load nested traces for a given project."""1148 is_root = None if load_nested else True1149 if isinstance(project, schemas.TracerSession):1150 runs = client.list_runs(project_id=project.id, is_root=is_root)1151 elif isinstance(project, uuid.UUID) or _is_uuid(project):1152 runs = client.list_runs(project_id=project, is_root=is_root)1153 else:1154 runs = client.list_runs(project_name=project, is_root=is_root)1155 if not load_nested:1156 return list(runs)1157 1158 treemap: collections.defaultdict[uuid.UUID, list[schemas.Run]] = (1159 collections.defaultdict(list)1160 )1161 results = []1162 all_runs = {}1163 for run in runs:1164 if run.parent_run_id is not None:1165 treemap[run.parent_run_id].append(run)1166 else:1167 results.append(run)1168 all_runs[run.id] = run1169 for run_id, child_runs in treemap.items():1170 all_runs[run_id].child_runs = sorted(child_runs, key=lambda r: r.dotted_order)1171 return results1172 1173 1174def _load_examples_map(1175 client: langsmith.Client, project: schemas.TracerSession1176) -> dict[uuid.UUID, schemas.Example]:1177 return {1178 e.id: e1179 for e in client.list_examples(1180 dataset_id=project.reference_dataset_id,1181 as_of=project.metadata.get("dataset_version"),1182 )1183 }1184 1185 1186IT = TypeVar("IT")1187 1188 1189def _load_tqdm() -> Callable[[IT], IT]:1190 try:1191 from tqdm.auto import tqdm1192 except ImportError:1193 return lambda x: x1194 return tqdm # type: ignore[return-value]1195 1196 1197ET = TypeVar("ET", bound="_ExperimentManagerMixin")1198 1199 1200class _ExperimentManagerMixin: