ó
    üÞ j"\ ã            	      ó†  • S r SSKJr  SSKrSSKJr  SSKrSSKrSSK	r	SSK
r
SSKrSSKrSSKrSSKrSSKrSSKrSSKJrJrJrJrJrJrJr  SSKJr  SSKJrJrJrJrJ r J!r!J"r"J#r#  SSK$J%r%J&r&  SSK'r'SSK'J(r)  SS	K'J*r+  SS
K'J,r-  SSK'J.r.  SSK'J/r0  SSK1J2r2  SSK3J4r4J5r5  SSK6J7r7J8r8J9r9J:r:J;r;J<r<J=r=J>r>J?r?  \(       a  SSK@rASSKBJCrC  \ARˆ                  rDO\rD\RŠ                  " \F5      rG\"\\H/\H4   \\H\H/\H4   4   rI\"\J\R–                  \\.R˜                     \.Rš                  4   rN\"\<\\.Rž                  \ \.R˜                     /\"\:\;4   4   \S\"\H\;\:4   4   4   rP\"\\.Rž                  \ \.R˜                     /\\"\:\;4      4   4   rQ\"\J\R–                  \.R¤                  4   rS\&            SG                             SHS jj5       rT\&            SG                             SIS jj5       rT             SJ                               SKS jjrT       SL                 SMS jjrU " S S\%5      rV " S S5      rW\\\.Rž                     \ \.R˜                     /\"\"\8\H4   \\"\8\H4      4   4   rX       SN                   SOS jjrY " S S5      rZ      SPS jr[    SQS  jr\SRS! jr]            SS                             STS" jjr^SUS# jr_      SVS$ jr` SW       SXS% jjra      SYS& jrb\!" S'5      rcSZS( jrd\!" S)S*S+9re " S, S*5      rf " S- S.\f5      rg    S[S/ jrh    S\S0 jri " S1 S2\%5      rj  S]                 S^S3 jjrkSUS4 jrl    S_S5 jrm      S`S6 jrnSS7.       SaS8 jjroSbS9 jrp    ScS: jrqSdS; jrrSeS< jrsSfS= jrtSgS> jruShS? jrv        SiS@ jrwSjSA jrx  Sk     SlSB jjry  Sk     SlSC jjrz\Rö                  " SSD9SmSE j5       r|SnSF jr}g)ozV2 Evaluation Interface.é    )ÚannotationsN)ÚAsyncIterableÚ	AwaitableÚ	GeneratorÚIterableÚIteratorÚSequenceÚSized)Úcopy_context)ÚTYPE_CHECKINGÚAnyÚCallableÚLiteralÚOptionalÚTypeVarÚUnionÚcast)Ú	TypedDictÚoverload)Úenv)Úrun_helpers)Ú	run_trees)Úschemas)Úutils)Ú_v2_migration_utils)Ú
_warn_onceÚsuppress_deprecation_warning)	ÚSUMMARY_EVALUATOR_TÚComparisonEvaluationResultÚDynamicComparisonRunEvaluatorÚEvaluationResultÚEvaluationResultsÚRunEvaluatorÚ_normalize_summary_evaluatorÚcomparison_evaluatorÚrun_evaluator©ÚRunnable.é   ÚExperimentResultsc               ó   • g ©N© ©ÚtargetÚdataÚ
evaluatorsÚsummary_evaluatorsÚmetadataÚexperiment_prefixÚdescriptionÚmax_concurrencyÚnum_repetitionsÚclientÚblockingÚ
experimentÚupload_resultsÚkwargss                 ÚV/var/www/html/gaurav/venv/lib/python3.13/site-packages/langsmith/evaluation/_runner.pyÚevaluater>   _   s   € ð" ó    ÚComparativeExperimentResultsc               ó   • g r,   r-   r.   s                 r=   r>   r>   s   s   € ð" $'r?   c               óØ  • [        U [        [        R                  [        R
                  45      (       aó  US:„  [        U5      U(       + [        U5      [        U5      S.n[        UR                  5       5      (       a/  S[        S UR                  5        5       5       S3n[        U5      e[        U [        [        R                  45      (       a  U OU R                  n[        R                  SU S35        [        U 4[!        ["        [$        [&              U5      UUUU	U
S.UD6$ [        U [(        [        45      (       GaA  US:„  [        U5      U(       + [        U5      [        U5      S	.n[+        U 5      S
:w  d  [-        S U  5       5      (       d  SU < 3n[        U5      e[        UR                  5       5      (       a/  S[        S UR                  5        5       5       S3n[        U5      eUb  X~S'   U  Vs/ sH6  n[        U[        [        R                  45      (       a  UOUR                  PM8     nn[        R                  SU S35        [/        U 4[!        [$        [0           U=(       d    S5      UUU	US.UD6$ U(       a  SU S3n[        U5      eU(       d  Sn[        U5      e[3        U 5      (       a(  [4        R6                  " U 5      (       a  Sn[        U5      eU(       a  U(       a  SU SU 3n[        U5      eU(       d  [9        S5        [        R                  SU  S35        [;        U U[!        ["        [$        [&              U5      UUUUUUU	U
UUUS9$ s  snf )a0  Evaluate a target system on a given dataset.

Args:
    target (TARGET_T | Runnable | EXPERIMENT_T | Tuple[EXPERIMENT_T, EXPERIMENT_T]):
        The target system or experiment(s) to evaluate.

        Can be a function that takes a dict and returns a `dict`, a langchain `Runnable`, an
        existing experiment ID, or a two-tuple of experiment IDs.
    data (DATA_T): The dataset to evaluate on.

        Can be a dataset name, a list of examples, or a generator of examples.
    evaluators (Sequence[EVALUATOR_T] | Sequence[COMPARATIVE_EVALUATOR_T] | None):
        A list of evaluators to run on each example. The evaluator signature
        depends on the target type.
    summary_evaluators (Sequence[SUMMARY_EVALUATOR_T] | None): A list of summary
        evaluators to run on the entire dataset.

        Should not be specified if comparing two existing experiments.
    metadata (dict | None): Metadata to attach to the experiment.
    experiment_prefix (str | None): A prefix to provide for your experiment name.
    description (str | None): A free-form text description for the experiment.
    max_concurrency (int | None): The maximum number of concurrent
        evaluations to run.

        If `None` then no limit is set. If `0` then no concurrency.
    client (langsmith.Client | None): The LangSmith client to use.
    blocking (bool): Whether to block until the evaluation is complete.
    num_repetitions (int): The number of times to run the evaluation.
        Each item in the dataset will be run and evaluated this many times.
    experiment (schemas.TracerSession | None): An existing experiment to
        extend.

        If provided, `experiment_prefix` is ignored.

        For advanced usage only. Should not be specified if target is an existing
        experiment or two-tuple fo experiments.
    error_handling (str, default="log"): How to handle individual run errors.

        `'log'` will trace the runs with the error message as part of the
        experiment, `'ignore'` will not count the run as part of the experiment at
        all.

Returns:
    ExperimentResults: If target is a function, `Runnable`, or existing experiment.
    ComparativeExperimentResults: If target is a two-tuple of existing experiments.

Examples:
    Prepare the dataset:

    >>> from typing import Sequence
    >>> from langsmith import Client
    >>> from langsmith.evaluation import evaluate
    >>> from langsmith.schemas import Example, Run
    >>> client = Client()
    >>> dataset = client.clone_public_dataset(
    ...     "https://smith.langchain.com/public/419dcab2-1d66-4b94-8901-0357ead390df/d"
    ... )
    >>> dataset_name = "Evaluate Examples"

    Basic usage:

    >>> def accuracy(run: Run, example: Example):
    ...     # Row-level evaluator for accuracy.
    ...     pred = run.outputs["output"]
    ...     expected = example.outputs["answer"]
    ...     return {"score": expected.lower() == pred.lower()}
    >>> def precision(runs: Sequence[Run], examples: Sequence[Example]):
    ...     # Experiment-level evaluator for precision.
    ...     # TP / (TP + FP)
    ...     predictions = [run.outputs["output"].lower() for run in runs]
    ...     expected = [example.outputs["answer"].lower() for example in examples]
    ...     # yes and no are the only possible answers
    ...     tp = sum([p == e for p, e in zip(predictions, expected) if p == "yes"])
    ...     fp = sum([p == "yes" and e == "no" for p, e in zip(predictions, expected)])
    ...     return {"score": tp / (tp + fp)}
    >>> def predict(inputs: dict) -> dict:
    ...     # This can be any function or just an API call to your app.
    ...     return {"output": "Yes"}
    >>> results = evaluate(
    ...     predict,
    ...     data=dataset_name,
    ...     evaluators=[accuracy],
    ...     summary_evaluators=[precision],
    ...     experiment_prefix="My Experiment",
    ...     description="Evaluating the accuracy of a simple prediction model.",
    ...     metadata={
    ...         "my-prompt-version": "abcd-1234",
    ...     },
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...

    Evaluating over only a subset of the examples

    >>> experiment_name = results.experiment_name
    >>> examples = client.list_examples(dataset_name=dataset_name, limit=5)
    >>> results = evaluate(
    ...     predict,
    ...     data=examples,
    ...     evaluators=[accuracy],
    ...     summary_evaluators=[precision],
    ...     experiment_prefix="My Experiment",
    ...     description="Just testing a subset synchronously.",
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...

    Streaming each prediction to more easily + eagerly debug.

    >>> results = evaluate(
    ...     predict,
    ...     data=dataset_name,
    ...     evaluators=[accuracy],
    ...     summary_evaluators=[precision],
    ...     description="I don't even have to block!",
    ...     blocking=False,
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...
    >>> for i, result in enumerate(results):  # doctest: +ELLIPSIS
    ...     pass



    Evaluating a LangChain object:

    >>> from langchain_core.runnables import chain as as_runnable
    >>> @as_runnable
    ... def nested_predict(inputs):
    ...     return {"output": "Yes"}
    >>> @as_runnable
    ... def lc_predict(inputs):
    ...     return nested_predict.invoke(inputs)
    >>> results = evaluate(
    ...     lc_predict.invoke,
    ...     data=dataset_name,
    ...     evaluators=[accuracy],
    ...     description="This time we're evaluating a LangChain object.",
    ...     summary_evaluators=[precision],
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...

!!! warning "Behavior changed in `langsmith` 0.2.0"

    'max_concurrency' default updated from None (no limit on concurrency)
    to 0 (no concurrency at all).
r)   )r7   r:   r;   r4   r0   zReceived invalid arguments. c              3  ó:   #   • U H  u  pU(       d  M  Uv •  M     g 7fr,   r-   ©Ú.0ÚkÚvs      r=   Ú	<genexpr>Úevaluate.<locals>.<genexpr>6  ó   é € ÐAÑ';™t˜q¼qŸ™Ò';ùó   ‚’	z? should not be specified when target is an existing experiment.z,Running evaluation over existing experiment z...)r1   r2   r3   r6   r8   r9   )r7   r:   r;   r2   r0   é   c              3  ó~   #   • U H4  n[        U[        [        R                  [        R
                  45      v •  M6     g 7fr,   )Ú
isinstanceÚstrÚuuidÚUUIDr   ÚTracerSession)rE   Úts     r=   rH   rI   N  s.   é € ð '
ÙLRÀqŒJ�qœ3¤§	¡	¬7×+@Ñ+@ÐA×BÐBÊFùs   ‚;=z­Received invalid target. If a tuple is specified it must have length 2 and each element should by the ID or schemas.TracerSession of an existing experiment. Received target=c              3  ó:   #   • U H  u  pU(       d  M  Uv •  M     g 7fr,   r-   rD   s      r=   rH   rI   Z  rJ   rK   zA should not be specified when target is two existing experiments.r6   z6Running pairwise evaluation over existing experiments r-   )r1   r4   r5   r8   r3   zReceived unsupported arguments zC. These arguments are not supported when creating a new experiment.zDMust specify 'data' when running evaluations over a target function.zåAsync functions are not supported by `evaluate`. Please use `aevaluate` instead:

from langsmith import aevaluate

await aevaluate(
    async_target_function,
    data=data,
    evaluators=evaluators,
    # ... other parameters
)zeExpected at most one of 'experiment' or 'experiment_prefix', but both were provided. Got: experiment=z, experiment_prefix=z&'upload_results' parameter is in beta.z&Running evaluation over target system )r0   r1   r2   r3   r4   r5   r6   r7   r8   r9   r:   r;   Úerror_handling)rN   rO   rP   rQ   r   rR   ÚboolÚanyÚvaluesÚtupleÚitemsÚ
ValueErrorÚidÚloggerÚdebugÚevaluate_existingr   r   r	   ÚEVALUATOR_TÚlistÚlenÚallÚevaluate_comparativeÚCOMPARATIVE_EVALUATOR_TÚcallableÚrhÚis_asyncr   Ú	_evaluate)r/   r0   r1   r2   r3   r4   r5   r6   r7   r8   r9   r:   r;   rU   r<   Úinvalid_argsÚmsgÚ	target_idrS   Ú
target_idss                       r=   r>   r>   ‡   s”  € ôH �&œ3¤§	¡	¬7×+@Ñ+@ÐA×BÑBà.°Ñ2Ü˜zÓ*Ø"0Ô0Ü!%Ð&7Ó!8Ü˜“Jñ
ˆô ˆ|×"Ñ"Ó$×%Ñ%à.ÜÑA |×'9Ñ'9Ô';ÓAÓAÐBð CCðDð ô
 ˜S“/Ð!Ü(¨´#´t·y±yÐ1A×BÑB‘FÈÏ	É	ˆ	Ü�‰ÐCÀIÀ;ÈcÐRÔSÜ Øð	
äœH¤X¬kÑ%:Ñ;¸ZÓHØ1ØØ+ØØñ	
ð ñ	
ð 		
ô 
�FœT¤5˜M×	*Ò	*à.°Ñ2Ü˜zÓ*Ø"0Ô0Ü"&Ð'9Ó":Ü˜“Jñ
ˆô ˆv‹;˜!Ó¤3ñ '
ÙLRó'
÷ $
ñ $
ð9à17±	ð;ð ô
 ˜S“/Ð!Ü�×$Ñ$Ó&×'Ñ'à.ÜÑA |×'9Ñ'9Ô';ÓAÓAÐBð CEðFð ô
 ˜S“/Ð!ØÑ&Ø(7Ð$Ñ%ÙNTÓUÉfÈœ: a¬#¬t¯y©yÐ)9×:Ñ:‘aÀÇÁÒDÉfˆ
ÐUÜ�‰ØDÀZÀLÐPSÐTô	
ô $Øð
äœHÔ%<Ñ=¸z×?OÈRÓPØ/Ø#ØØñ
ð ñ
ð 	
ö 
à-¨f¨Xð 68ð 9ð 	ô ˜‹oÐÞØTˆÜ˜‹oÐÜ	�&×	Ñ	œbŸkšk¨&×1Ñ1ðð 	ô ˜‹oÐÞ	Ö)ðà)˜lÐ*>Ð?PÐ>QðSð 	ô
 ˜‹oÐæÜÐ?Ô@Ü�‰Ð=¸f¸XÀSÐIÔJÜØØÜœH¤X¬kÑ%:Ñ;¸ZÓHØ1ØØ/Ø#Ø+Ø+ØØØ!Ø)Ø)ñ
ð 	
ùò] Vs   Ç><M'Fc               ó  • U=(       d    [         R                  " SS9n[        X5      n[        X…US9n	[	        XX5      n
U	 Vs/ sH)  oº[        [        R                  UR                  5         PM+     nn[        U	UUUUUUUUS9	$ s  snf )aã  Evaluate existing experiment runs.

Args:
    experiment (Union[str, uuid.UUID]): The identifier of the experiment to evaluate.
    evaluators (Optional[Sequence[EVALUATOR_T]]): Optional sequence of evaluators to use for individual run evaluation.
    summary_evaluators (Optional[Sequence[SUMMARY_EVALUATOR_T]]): Optional sequence of evaluators
        to apply over the entire dataset.
    metadata (Optional[dict]): Optional metadata to include in the evaluation results.
    max_concurrency (int | None): The maximum number of concurrent
        evaluations to run.

        If `None` then no limit is set. If `0` then no concurrency.
    client (Optional[langsmith.Client]): Optional Langsmith client to use for evaluation.
    load_nested: Whether to load all child runs for the experiment.

        Default is to only load the top-level root runs.
    blocking (bool): Whether to block until evaluation is complete.

Returns:
    The evaluation results.

Environment:
    - `LANGSMITH_TEST_CACHE`: If set, API calls will be cached to disk to save time and
        cost during testing.

        Recommended to commit the cache files to your repository for faster CI/CD runs.

        Requires the `'langsmith[vcr]'` package to be installed.

Examples:
    Define your evaluators

    >>> from typing import Sequence
    >>> from langsmith.schemas import Example, Run
    >>> def accuracy(run: Run, example: Example):
    ...     # Row-level evaluator for accuracy.
    ...     pred = run.outputs["output"]
    ...     expected = example.outputs["answer"]
    ...     return {"score": expected.lower() == pred.lower()}
    >>> def precision(runs: Sequence[Run], examples: Sequence[Example]):
    ...     # Experiment-level evaluator for precision.
    ...     # TP / (TP + FP)
    ...     predictions = [run.outputs["output"].lower() for run in runs]
    ...     expected = [example.outputs["answer"].lower() for example in examples]
    ...     # yes and no are the only possible answers
    ...     tp = sum([p == e for p, e in zip(predictions, expected) if p == "yes"])
    ...     fp = sum([p == "yes" and e == "no" for p, e in zip(predictions, expected)])
    ...     return {"score": tp / (tp + fp)}

    Load the experiment and run the evaluation.

    >>> import uuid
    >>> from langsmith import Client
    >>> from langsmith.evaluation import evaluate, evaluate_existing
    >>> client = Client()
    >>> dataset_name = "__doctest_evaluate_existing_" + uuid.uuid4().hex[:8]
    >>> dataset = client.create_dataset(dataset_name)
    >>> example = client.create_example(
    ...     inputs={"question": "What is 2+2?"},
    ...     outputs={"answer": "4"},
    ...     dataset_id=dataset.id,
    ... )
    >>> def predict(inputs: dict) -> dict:
    ...     return {"output": "4"}
    >>> # First run inference on the dataset
    ... results = evaluate(
    ...     predict, data=dataset_name, experiment_prefix="doctest_experiment"
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...
    >>> experiment_id = results.experiment_name
    >>> # Wait for the experiment to be fully processed and check if we have results
    >>> len(results) > 0
    True
    >>> import time
    >>> time.sleep(5)  # Wait longer for runs to be indexed
    >>> results = evaluate_existing(
    ...     experiment_id,
    ...     evaluators=[accuracy],
    ...     summary_evaluators=[precision],
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...
    >>> client.delete_dataset(dataset_id=dataset.id)
)i N  i‘_ )Ú
timeout_ms©Úload_nested)r0   r1   r2   r3   r6   r8   r9   r:   )
ÚrtÚget_cached_clientÚ_load_experimentÚ_load_traces_for_experimentÚ_load_examples_mapr   rP   rQ   Úreference_example_idri   )r:   r1   r2   r3   r6   r8   rq   r9   ÚprojectÚrunsÚdata_mapÚrunr0   s                r=   r_   r_      s‘   € ð| ×H”r×+Ò+Ð7GÑH€FÜ˜zÓ2€GÜ& wÀKÑP€DÜ! &Ó2€HÙKOÓPÉ4ÀC”Tœ$Ÿ)™) S×%=Ñ%=Ó>Ô?É4€DÐPÜØØØØ-ØØ'ØØØñ
ð 
ùò Qs   Á/Bc                  ó4   • \ rS rSr% S\S'   S\S'   S\S'   Srg	)
ÚExperimentResultRowi  úschemas.Runr{   úschemas.ExampleÚexampler"   Úevaluation_resultsr-   N©Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__annotations__Ú__static_attributes__r-   r?   r=   r}   r}     s   ‡ Ø	ÓØÓØ)Ö)r?   r}   c                  óÖ   • \ rS rSrSrSSS jjr\SS j5       r\SS j5       r\SS j5       r	SS jr
\SS j5       rSS	 jrSS
 jrSS jr S     SS jjrSS jrSS jrSS jrSrg)r*   i  a¥  Represents the results of an evaluate() call.

This class provides an iterator interface to iterate over the experiment results
as they become available. It also provides methods to access the experiment name,
the number of results, and to wait for the results to be processed.

Methods:
    experiment_name() -> str: Returns the name of the experiment.
    wait() -> None: Waits for the experiment data to be processed.
c                óL  • Xl         / U l        [        R                  " 5       U l        [
        R                  " 5       U l        S U l        U(       d>  [
        R                  " U R                  S9U l        U R                  R                  5         g S U l        U R                  5         g )N©r/   )Ú_managerÚ_resultsÚqueueÚQueueÚ_queueÚ	threadingÚEventÚ_processing_completeÚ_processing_errorÚThreadÚ_process_dataÚ_threadÚstart)ÚselfÚexperiment_managerr9   s      r=   Ú__init__ÚExperimentResults.__init__"  su   € Ø*ŒØ35ˆŒÜ8=¿º»ˆŒÜ$-§O¢OÓ$5ˆÔ!Ø:>ˆÔÞÜ7@×7GÒ7GØ×)Ñ)ñ8ˆDŒLð �L‰L×ÑÕ àˆDŒLØ×ÑÕ r?   c                ó.   • U R                   R                  $ r,   )rŒ   Úexperiment_name©r™   s    r=   rž   Ú!ExperimentResults.experiment_name1  s   € à�}‰}×,Ñ,Ð,r?   c                óJ   • U R                   R                  5       R                  $ )zThe ID of the experiment.)rŒ   Ú_get_experimentr\   rŸ   s    r=   Úexperiment_idÚExperimentResults.experiment_id5  s   € ð �}‰}×,Ñ,Ó.×1Ñ1Ð1r?   c                ó  • U R                   R                  5       nUR                  (       aZ  UR                  R                  S5      S   nUR                  S5      S   nU SU R                   R                   SUR
                   3$ g)z.The URL of the experiment in the LangSmith UI.Ú?r   ú/projects/p/ú
/datasets/ú/compare?selectedSessions=N)rŒ   r¢   ÚurlÚsplitÚ
dataset_idr\   )r™   r:   Úproject_urlÚbase_urls       r=   rª   ÚExperimentResults.url:  s   € ð —]‘]×2Ñ2Ó4ˆ
Ø�>�>Ø$Ÿ.™.×.Ñ.¨sÓ3°AÑ6ˆKØ"×(Ñ(¨Ó8¸Ñ;ˆHà�*˜J t§}¡}×'?Ñ'?Ð&@ð A$Ø$.§M¡M ?ð4ðð r?   c                ó.   • U R                   R                  $ )z:Get the ID of the dataset associated with this experiment.)rŒ   r¬   rŸ   s    r=   Úget_dataset_idÚ ExperimentResults.get_dataset_idG  s   € à�}‰}×'Ñ'Ð'r?   c                ó   • U R                   $ )z3The URL to the comparison view for this experiment.)rª   rŸ   s    r=   Úcomparison_urlÚ ExperimentResults.comparison_urlK  s   € ð �x‰xˆr?   c              #  ó¬  #   • SnU R                   R                  5       (       a8  U R                  R                  5       (       a  U[	        U R
                  5      :  a©   U[	        U R
                  5      :  a  U R
                  U   v •  US-  nOU R                  R                  SSS9   U R                   R                  5       (       d  Mm  U R                  R                  5       (       d  MŽ  U[	        U R
                  5      :  a  M©  U R                  b  U R                  eg ! [        R                   a    U R                  b  U R                  e GMK  f = f7f)Nr   r)   Tgš™™™™™¹?)ÚblockÚtimeout)
r“   Úis_setr�   Úemptyrb   r�   ÚgetrŽ   ÚEmptyr”   )r™   Úixs     r=   Ú__iter__ÚExperimentResults.__iter__P  s  é € Øˆà×)Ñ)×0Ñ0×2Ñ2Ø—;‘;×$Ñ$×&Ñ&Ø”C˜Ÿ™Ó&Ó&ð	Øœ˜DŸM™MÓ*Ó*ØŸ-™-¨Ñ+Ò+Ø˜!‘G‘Bà—K‘K—O‘O¨$¸�OÒ<ð ×)Ñ)×0Ñ0×2Ó2Ø—;‘;×$Ñ$×&Ó&Ø”C˜Ÿ™Ó&Õ&ð ×!Ñ!Ñ-Ø×(Ñ(Ð(ð .øô	 —;‘;ó Ø×)Ñ)Ñ5Ø×0Ñ0Ð0ÛðüsH   ‚AEÁ/D ÂEÂD Â' EÃ	EÃ*EÄEÄ-EÅEÅEÅEc                óÄ  • [        5       n U R                  R                  5       nU" U5       H9  nU R                  R	                  U5        U R
                  R                  U5        M;     U R                  R                  5       nX@l        U R                  R                  5         g ! [         a  nXPl
         S nAN0S nAff = f! U R                  R                  5         f = fr,   )Ú
_load_tqdmrŒ   Úget_resultsr�   Úputr�   ÚappendÚget_summary_scoresÚ_summary_resultsÚBaseExceptionr”   r“   Úset)r™   ÚtqdmÚresultsÚitemÚsummary_scoresÚes         r=   r–   ÚExperimentResults._process_datad  s®   € Ü‹|ˆð	,Ø—m‘m×/Ñ/Ó1ˆGÙ˜Wž�Ø—‘—‘ Ô%Ø—‘×$Ñ$ TÖ*ñ &ð "Ÿ]™]×=Ñ=Ó?ˆNØ$2Ô!ð ×%Ñ%×)Ñ)Õ+øô ó 	'Ø%&×"Ñ"ûð	'ûð ×%Ñ%×)Ñ)Õ+ús*   ŒA?B& Â&
C Â0B;Â6C Â;C Ã C ÃCc                ó,   • [        U R                  5      $ r,   )rb   r�   rŸ   s    r=   Ú__len__ÚExperimentResults.__len__s  s   € Ü�4—=‘=Ó!Ð!r?   Nc                ó*   • [        U R                  XS9$ )N©r˜   Úend)Ú
_to_pandasr�   )r™   r˜   rÔ   s      r=   Ú	to_pandasÚExperimentResults.to_pandasv  s   € ô ˜$Ÿ-™-¨uÑ>Ð>r?   c                óÌ   • SS K nU R                  (       a@  UR                  R                  S5      (       a   U R	                  5       nUR                  5       $ U R                  5       $ )Nr   Úpandas)Úimportlib.utilr�   ÚutilÚ	find_specrÖ   Ú_repr_html_Ú__repr__)r™   Ú	importlibÚdfs      r=   rÝ   ÚExperimentResults._repr_html_{  sE   € Ûà�=�=˜YŸ^™^×5Ñ5°h×?Ñ?Ø—‘Ó!ˆBØ—>‘>Ó#Ð#à—=‘=“?Ð"r?   c                ó"   • SU R                    S3$ )Nz<ExperimentResults Ú>)rž   rŸ   s    r=   rÞ   ÚExperimentResults.__repr__„  s   € Ø$ T×%9Ñ%9Ð$:¸!Ð<Ð<r?   c                óŒ   • U R                   (       a  U R                   R                  5         U R                  b  U R                  eg)z‹Wait for the evaluation runner to complete.

This method blocks the current thread until the evaluation runner has
finished its execution.
N)r—   Újoinr”   rŸ   s    r=   ÚwaitÚExperimentResults.wait‡  s8   € ð �<�<Ø�L‰L×ÑÔØ×!Ñ!Ñ-Ø×(Ñ(Ð(ð .r?   )rŒ   r“   r”   r�   r�   rÆ   r—   )T)rš   Ú_ExperimentManagerr9   rV   ©ÚreturnrO   )rë   ú	uuid.UUID©rë   úOptional[str])rë   zIterator[ExperimentResultRow]©rë   ÚNone)rë   Úint©r   N)r˜   úOptional[int]rÔ   ró   rë   Ú	DataFrame)rƒ   r„   r…   r†   Ú__doc__r›   Úpropertyrž   r£   rª   r±   r´   r¾   r–   rÐ   rÖ   rÝ   rÞ   rç   rˆ   r-   r?   r=   r*   r*     s¢   † ñ	ö!ð ó-ó ð-ð ó2ó ð2ð ó
ó ð
ô(ð óó ðô)ô(,ô"ð >Bð?Ø"ð?Ø-:ð?à	õ?ô
#ô=÷	)r?   c	          
     ó¬	  ^^^.• [        U 5      S:  a  [        S5      eU(       d  [        S5      eUS:  a  [        S5      eT=(       d    [        R                  " 5       mU  V	s/ sH  n	[	        U	T5      PM     n
n	U
 Vs/ sH  n[        UR                  5      PM     nn[        [        U5      5      S:X  d  [        S5      eU
 Vs/ sH  o»R                  PM     nnUcj  U
 Vs/ sH  o»R                  c  M  UR                  PM     nnS	R                  U5      S
-   [        [        R                  " 5       R                  SS 5      -   nO1US
-   [        [        R                  " 5       R                  SS 5      -   n[        R                  " 5       nTR                  UUUUUS9m.[        [         ["        R$                  ["        R$                  4   [!        U
5      5      n['        UT.5      n[)        U5        U
 Vs/ sH  n[+        UTUS9PM     nnSnU H*  nU Vs1 sH  nUR,                  iM     nnUc  UnM%  UU-  nM,     Ub  [/        U5      O/ nU Vs/ sH
  nUc  M  UPM     nnSn0 n[1        S[        U5      U5       H[  nUUUU-    nTR3                  U
S   R                  U
S   R4                  R7                  S5      US9 H  n U UU R                  '   M     M]     [8        R:                  " [.        5      n!U HT  nU HK  nUR,                  U;   d  M  U![        [        R<                  UR,                  5         R?                  U5        MM     MV     U=(       d    /  V"s/ sH  n"[A        U"5      PM     n#n"0 n$          SUU.U4S jjn%[C        5       n&[D        RF                  " U=(       d    SS9 n'/ n(U&" U!RI                  5       5       Hm  u  n)nSU0U$U)'   U# HZ  n*US:”  a+  U'RK                  U%UUU)   U*U'5      n+U(R?                  U+5        M4  U%" UUU)   U*U'5      u  n,n-U-U$U)   SU-RL                   3'   M\     Mo     U((       aG  [N        RP                  " U(5        U( H+  n+U+RS                  5       u  n)n-U-U$U)   SU-RL                   3'   M-     SSS5        [U        U$UT.US9$ s  sn	f s  snf s  snf s  snf s  snf s  snf s  snf s  sn"f ! , (       d  f       NB= f)aÊ  Evaluate existing experiment runs against each other.

This lets you use pairwise preference scoring to generate more
reliable feedback in your experiments.

Args:
    experiments (Tuple[Union[str, uuid.UUID], Union[str, uuid.UUID]]):
        The identifiers of the experiments to compare.
    evaluators (Sequence[COMPARATIVE_EVALUATOR_T]):
        A list of evaluators to run on each example.
    experiment_prefix (Optional[str]): A prefix to provide for your experiment name.
    description (Optional[str]): A free-form text description for the experiment.
    max_concurrency (int): The maximum number of concurrent evaluations to run.
    client (Optional[langsmith.Client]): The LangSmith client to use.
    metadata (Optional[dict]): Metadata to attach to the experiment.
    load_nested (bool): Whether to load all child runs for the experiment.

        Default is to only load the top-level root runs.
    randomize_order (bool): Whether to randomize the order of the outputs for each evaluation.

Returns:
    The results of the comparative evaluation.

Examples:
    Suppose you want to compare two prompts to see which one is more effective.
    You would first prepare your dataset:

    >>> from typing import Sequence
    >>> from langsmith import Client
    >>> from langsmith.evaluation import evaluate
    >>> from langsmith.schemas import Example, Run
    >>> client = Client()
    >>> dataset = client.clone_public_dataset(
    ...     "https://smith.langchain.com/public/419dcab2-1d66-4b94-8901-0357ead390df/d"
    ... )
    >>> dataset_name = "Evaluate Examples"

    Then you would run your different prompts:
    >>> import functools
    >>> import openai
    >>> from langsmith.evaluation import evaluate
    >>> from langsmith.wrappers import wrap_openai
    >>> oai_client = openai.Client()
    >>> wrapped_client = wrap_openai(oai_client)
    >>> prompt_1 = "You are a helpful assistant."
    >>> prompt_2 = "You are an exceedingly helpful assistant."
    >>> def predict(inputs: dict, prompt: str) -> dict:
    ...     completion = wrapped_client.chat.completions.create(
    ...         model="gpt-4o-mini",
    ...         messages=[
    ...             {"role": "system", "content": prompt},
    ...             {
    ...                 "role": "user",
    ...                 "content": f"Context: {inputs['context']}"
    ...                 f"\n\ninputs['question']",
    ...             },
    ...         ],
    ...     )
    ...     return {"output": completion.choices[0].message.content}
    >>> results_1 = evaluate(
    ...     functools.partial(predict, prompt=prompt_1),
    ...     data=dataset_name,
    ...     description="Evaluating our basic system prompt.",
    ...     blocking=False,  # Run these experiments in parallel
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...
    >>> results_2 = evaluate(
    ...     functools.partial(predict, prompt=prompt_2),
    ...     data=dataset_name,
    ...     description="Evaluating our advanced system prompt.",
    ...     blocking=False,
    ... )  # doctest: +ELLIPSIS
    View the evaluation results for experiment:...
    >>> results_1.wait()
    >>> results_2.wait()

        Finally, you would compare the two prompts directly:
    >>> import json
    >>> from langsmith.evaluation import evaluate_comparative
    >>> from langsmith import schemas
    >>> def score_preferences(runs: list, example: schemas.Example):
    ...     assert len(runs) == 2  # Comparing 2 systems
    ...     assert isinstance(example, schemas.Example)
    ...     assert all(run.reference_example_id == example.id for run in runs)
    ...     pred_a = runs[0].outputs["output"] if runs[0].outputs else ""
    ...     pred_b = runs[1].outputs["output"] if runs[1].outputs else ""
    ...     ground_truth = example.outputs["answer"] if example.outputs else ""
    ...     tools = [
    ...         {
    ...             "type": "function",
    ...             "function": {
    ...                 "name": "rank_preferences",
    ...                 "description": "Saves the prefered response ('A' or 'B')",
    ...                 "parameters": {
    ...                     "type": "object",
    ...                     "properties": {
    ...                         "reasoning": {
    ...                             "type": "string",
    ...                             "description": "The reasoning behind the choice.",
    ...                         },
    ...                         "preferred_option": {
    ...                             "type": "string",
    ...                             "enum": ["A", "B"],
    ...                             "description": "The preferred option, either 'A' or 'B'",
    ...                         },
    ...                     },
    ...                     "required": ["preferred_option"],
    ...                 },
    ...             },
    ...         }
    ...     ]
    ...     completion = openai.Client().chat.completions.create(
    ...         model="gpt-4o-mini",
    ...         messages=[
    ...             {"role": "system", "content": "Select the better response."},
    ...             {
    ...                 "role": "user",
    ...                 "content": f"Option A: {pred_a}"
    ...                 f"\n\nOption B: {pred_b}"
    ...                 f"\n\nGround Truth: {ground_truth}",
    ...             },
    ...         ],
    ...         tools=tools,
    ...         tool_choice={
    ...             "type": "function",
    ...             "function": {"name": "rank_preferences"},
    ...         },
    ...     )
    ...     tool_args = completion.choices[0].message.tool_calls[0].function.arguments
    ...     loaded_args = json.loads(tool_args)
    ...     preference = loaded_args["preferred_option"]
    ...     comment = loaded_args["reasoning"]
    ...     if preference == "A":
    ...         return {
    ...             "key": "ranked_preference",
    ...             "scores": {runs[0].id: 1, runs[1].id: 0},
    ...             "comment": comment,
    ...         }
    ...     else:
    ...         return {
    ...             "key": "ranked_preference",
    ...             "scores": {runs[0].id: 0, runs[1].id: 1},
    ...             "comment": comment,
    ...         }
    >>> def score_length_difference(runs: list, example: schemas.Example):
    ...     # Just return whichever response is longer.
    ...     # Just an example, not actually useful in real life.
    ...     assert len(runs) == 2  # Comparing 2 systems
    ...     assert isinstance(example, schemas.Example)
    ...     assert all(run.reference_example_id == example.id for run in runs)
    ...     pred_a = runs[0].outputs["output"] if runs[0].outputs else ""
    ...     pred_b = runs[1].outputs["output"] if runs[1].outputs else ""
    ...     if len(pred_a) > len(pred_b):
    ...         return {
    ...             "key": "length_difference",
    ...             "scores": {runs[0].id: 1, runs[1].id: 0},
    ...         }
    ...     else:
    ...         return {
    ...             "key": "length_difference",
    ...             "scores": {runs[0].id: 0, runs[1].id: 1},
    ...         }
    >>> results = evaluate_comparative(
    ...     [results_1.experiment_name, results_2.experiment_name],
    ...     evaluators=[score_preferences, score_length_difference],
    ...     client=client,
    ... )  # doctest: +ELLIPSIS
    View the pairwise evaluation results at:...
    >>> eval_results = list(results)
    >>> assert len(eval_results) >= 10  # doctest: +SKIP
    >>> assert all(
    ...     "feedback.ranked_preference" in r["evaluation_results"]
    ...     for r in eval_results
    ... )  # doctest: +SKIP
    >>> assert all(
    ...     "feedback.length_difference" in r["evaluation_results"]
    ...     for r in eval_results
    ... )  # doctest: +SKIP
rL   z7Comparative evaluation requires at least 2 experiments.z>At least one evaluator is required for comparative evaluation.r   z+max_concurrency must be a positive integer.r)   z5All experiments must have the same reference dataset.Nz vs. Ú-é   é   )Úexperimentsr5   r3   r\   rp   éc   Údataset_version)r¬   Úas_ofÚexample_idsc                óÀ  >• [         R                  " 5       nT(       a  [        R                  " U 5        [        R
                  " STS9   UR                  X5      nTc  [        S5      e S S S 5        [        WR                  [        5      (       a0  UR                   Vs0 sH  n[        U5      UR                  _M     snOUR                  =(       d    0 nU  Vs0 sH  n[        UR                  5      U_M     n	nUR                  R                  5        H�  u  p«U	R                  [        U
5      5      nUR                  TR                   U
UR"                  UUR                  [        U
5      5      TR                  UR$                  UU(       a  UR&                  OS U(       a  UR(                  OS S9
  MŸ     UR                  U4$ ! , (       d  f       GNa= fs  snf s  snf )Nr1   )Úproject_namer8   z&Client is required to submit feedback.)	Úrun_idÚkeyÚscoreÚcommentÚcomparative_experiment_idÚsource_run_idÚfeedback_group_idÚ
session_idÚ
start_time)rP   Úuuid4ÚrandomÚshufflerg   Útracing_contextÚcompare_runsr[   rN   r  rO   Úscoresr\   rZ   r»   ÚsubmitÚcreate_feedbackr  r  r	  r
  )Ú	runs_listr€   Ú
comparatorÚexecutorr  ÚresultÚridÚcommentsr{   Ú
runs_by_idr  r  r8   Úcomparative_experimentÚrandomize_orders               €€€r=   Úevaluate_and_submit_feedbackÚ:evaluate_comparative.<locals>.evaluate_and_submit_feedback¨  sz  ø€ ô !ŸJšJ›LÐÞÜ�NŠN˜9Ô%Ü×Ò¨\À&ÓIØ×,Ñ,¨YÓ@ˆFØ‰~Ü Ð!IÓJÐJð ÷ Jô ˜&Ÿ.™.¬#×.Ñ.ð 28·²Ó?±¨#ŒS�‹X�v—~‘~Ò%±Ò?à—.‘.×& Bð 	ñ 3<Ó<±)¨3”c˜#Ÿ&™&“k 3Ò&±)ˆ
Ð<Ø#Ÿ]™]×0Ñ0Ö2‰MˆFØ—.‘.¤ V£Ó-ˆCØ�O‰OØ×&Ñ&ØØ—J‘JØØ Ÿ™¤S¨£[Ó1Ø*@×*CÑ*CØ$×2Ñ2Ø"3Þ-0˜3Ÿ>š>°dÞ-0˜3Ÿ>š>°dð ó ñ 3ð �z‰z˜6Ð!Ð!÷3 JÖIüò
 @ùò
 =s   Á	 GÂGÃGÇ
G©Úmax_workersry   ú	feedback.)r  rª   )
r  úlist[schemas.Run]r€   r   r  r    r  zcf.Executorrë   z,tuple[uuid.UUID, ComparisonEvaluationResult])+rb   r[   rr   rs   rt   rO   Úreference_dataset_idrÈ   r\   Únameræ   rP   r  ÚhexÚcreate_comparative_experimentr   rY   r   ÚTracerSessionResultÚ_build_comparative_urlÚ#_print_comparative_experiment_startru   rw   ra   ÚrangeÚlist_examplesr3   r»   ÚcollectionsÚdefaultdictrQ   rÄ   r%   rÁ   Úls_utilsÚContextThreadPoolExecutorrZ   r  r  Úcfrç   r  r@   )/rû   r1   r4   r5   r6   r8   r3   rq   r  r:   ÚprojectsÚpÚref_datasets_Úexperiment_idsÚexperiment_namesrž   r  Úexperiments_tupler´   rx   ry   Úexamples_intersectionr  r{   Úexample_ids_setÚexample_ids_nullableÚeidrÿ   Ú
batch_sizer0   ÚiÚexample_ids_batchrÍ   Ú	runs_dictÚ	evaluatorÚcomparatorsrÊ   r  rÉ   r  ÚfuturesÚ
example_idr  ÚfutureÚ_r  r  s/        `  `                                     @r=   rd   rd   Ÿ  s  ú€ ô~ ˆ;Ó˜!ÓÜÐRÓSÐSÞÜØLó
ð 	
ð ˜ÓÜÐFÓGÐGØ×-”r×+Ò+Ó-€Fñ HSÓSÁ{¸Ô  ¨VÖ4Á{€HÐSÙ:BÓC¹(°Q”S˜×/Ñ/Ö0¹(€MÐCÜŒs�=Ó!Ó" aÓ'ÜÐPÓQÐQÙ$,Ó-¡H˜q—d”d¡H€NÐ-ØÑ Ù,4ÓK©H q¿¹›F˜AŸFœF©HÐÐKà�L‰LÐ)Ó*¨SÑ0´3´t·z²z³|×7GÑ7GÈÈÐ7KÓ3LÑLñ 	ð ,¨cÑ1´C¼¿
º
»×8HÑ8HÈÈ!Ð8LÓ4MÑMˆÜ $§
¢
£ÐØ#×AÑAØØ"ØØØ$ð Bð Ðô ÜŒg×)Ñ)¬7×+FÑ+FÐFÑGÜˆh‹óÐô ,Ð,=Ð?UÓV€NÜ'¨Ô7ñ  óáˆGô 	$ G¨VÀÔMÙð 	ð ð
 !ÐÛˆ	Ù?HÓI¹y¸˜3×3Ô3¹yˆÐIØ Ñ(Ø$3Ò!à! _Ñ4Ò!ñ ð (=Ñ'HŒÐ"Ô#Èbð ñ #7ÓJÑ"6˜3¸#—3Ñ"6€KÐJð €JØ€DÜ�1”c˜+Ó&¨
Ö3ˆØ'¨¨A°
©NÐ;ÐØ×%Ñ%Ø ‘{×7Ñ7Ø˜1‘+×&Ñ&×*Ñ*Ð+<Ó=Ø)ð &ó 
ˆAð
 ˆD�—‘‹Jó
ñ 4ô 5@×4KÒ4KÌDÓ4Q€IÛˆ	ÛˆCØ×'Ñ'¨4Õ/Øœ$œtŸy™y¨#×*BÑ*BÓCÑD×KÑKÈCÖPó ñ ð
 EO×DTÐRTÐDTÓUÑDT°yÔ'¨	Ö2ÑDT€KÐUØ€Gð""Ø$ð""à ð""ð 2ð""ð ð	""ð
 
6÷""ñ ""ôH ‹<€DÜ	×	+Ò	+Ø#×( qò
à	ØˆÙ%)¨)¯/©/Ó*;Ö%<Ñ!ˆJ˜	Ø#)¨9Ð"5ˆG�JÑÛ)�
Ø" QÓ&Ø%Ÿ_™_Ø4Ø!Ø˜ZÑ(Ø"Ø ó�Fð —N‘N 6Ö*á <Ø! 4¨
Ñ#3°ZÀó!‘I�A�vð EK�G˜JÑ'¨)°F·J±J°<Ð(@ÓAó *ñ &=ö" Ü�GŠG�GÔÛ!�Ø%+§]¡]£_Ñ"�
˜FØ@F�˜
Ñ# i°·
±
¨|Ð$<Ó=ñ "÷/
ô6 (ØØØ5Øñ	ð ùò TùÚCùò .ùâKùò(ùò Jùò Kùò& V÷P
õ 
úsI   Á#RÁ>R"ÃR'Ã"R,Ã5R,Ç6R1ÈR6ÉR;ÉR;ÍS Î1CSÓ
Sc                  ón   • \ rS rSrSr   S
       SS jjr\SS j5       r\SS j5       rS r	S r
S	rg)r@   ið  aø  Represents the results of an evaluate_comparative() call.

This class provides an iterator interface to iterate over the comparison results,
indexed access by example ID, and properties to access the pairwise comparison
URL and the underlying comparative experiment.

Attributes:
    url (Optional[str]): URL of the pairwise comparison view in the LangSmith UI.
    comparative_experiment (Optional[schemas.ComparativeExperiment]): The
    comparative experiment, exposing its id, dataset, and metadata.
Nc                ó4   • Xl         X l        X0l        X@l        g r,   )r�   Ú	_examplesÚ_comparative_experimentÚ_url)r™   rÊ   Úexamplesr  rª   s        r=   r›   Ú%ComparativeExperimentResults.__init__ý  s   € ð  ŒØ!ŒØ'=Ô$Ø�	r?   c                ó   • U R                   $ )zCThe comparative experiment, exposing its id, dataset, and metadata.)rG  rŸ   s    r=   r  Ú3ComparativeExperimentResults.comparative_experiment	  s   € ð ×+Ñ+Ð+r?   c                ó   • U R                   $ )z8URL of the pairwise comparison view in the LangSmith UI.)rH  rŸ   s    r=   rª   Ú ComparativeExperimentResults.url  s   € ð �y‰yÐr?   c                ó    • U R                   U   $ )z0Return the result associated with the given key.)r�   )r™   r  s     r=   Ú__getitem__Ú(ComparativeExperimentResults.__getitem__  s   € à�}‰}˜SÑ!Ð!r?   c              #  ó    #   • U R                   R                  5        H,  u  pU R                  (       a  U R                  U   OS US.v •  M.     g 7f)N)r€   r�   )r�   rZ   rF  )r™   r  Úvalues      r=   r¾   Ú%ComparativeExperimentResults.__iter__  s?   é € ØŸ-™-×-Ñ-Ö/‰JˆCà26·.·.˜4Ÿ>™>¨#Ò.ÀdØ&+ñô ò 0ùs   ‚AA)rG  rF  r�   rH  ©NNN)rÊ   ÚdictrI  z*Optional[dict[uuid.UUID, schemas.Example]]r  ú'Optional[schemas.ComparativeExperiment]rª   rî   )rë   rW  rí   )rƒ   r„   r…   r†   rõ   r›   rö   r  rª   rP  r¾   rˆ   r-   r?   r=   r@   r@   ð  sr   † ñ
ð @DØJNØ!ð
àð
ð =ð
ð !Hð	
ð
 õ
ð ó,ó ð,ð óó ðò"õr?   c                ó`  • U S   R                   =(       d    U S   R                   nU(       d  g UR                  S5      S   nUR                  nUR                  S5      S   nU SU SSR                  U  Vs/ sH  n[	        UR
                  5      PM     sn5       SUR
                   3$ s  snf )	Nr   r)   r¦   r§   r¨   r©   z%2Cz&comparativeExperiment=)rª   r«   r"  ræ   rO   r\   )rû   r  rª   r­   r¬   r®   rÍ   s          r=   r'  r'  "  s¬   € ð �a‰.×
Ñ
×
2 ¨A¡× 2Ñ 2€CÞØØ—)‘)˜C“. Ñ#€KØ'×<Ñ<€JØ× Ñ  Ó0°Ñ3€Hàˆ*�J˜z˜lð +Ø!ŸJ™J¹;Ó'G¹;°a¬¨A¯D©D®	¹;Ñ'GÓHÐIØ
!Ð"8×";Ñ";Ð!<ð	>ðùâ'Gs   Á7B+c                ó2   • U (       a  [        SU  S35        g g )Nz)View the pairwise evaluation results at:
ú

)Úprint)r´   s    r=   r(  r(  3  s!   € ö ÜØ8¸Ð8HÈÐMõ	
ð r?   c                ó<   • [        U 5      =(       d    [        U 5      $ r,   )rf   Ú_is_langchain_runnabler‹   s    r=   Ú_is_callabler^  <  s   € Ü�FÓ×=Ô5°fÓ=Ð=r?   c               ó
  • U	=(       d    [         R                  " 5       n	[        U 5      (       a  S O [        [        [
        R                     U 5      n[        X¾U	5      u  pþ[        UU	UU=(       d    UUU[        U5      U[        X5      UUS9R                  5       n[        R                  " S 5      =n(       a'  [        R                  " U5      UR                    S3-  nOS n[        R"                  " UU	R$                  /S9   [        U 5      (       a  UR'                  [        [(        U 5      US9nU(       a  UR+                  X'S9nU(       a  UR-                  U5      n[/        UU
S9nUsS S S 5        $ ! , (       d  f       g = f)N)
r8   r3   r:   r5   r7   Úevaluator_keysry   Úinclude_attachmentsr;   rU   z.yaml)Úignore_hosts©r6   )r9   )rr   rs   r^  r   r   r   ÚRunÚ_resolve_experimentré   Ú_collect_evaluator_keysÚ_include_attachmentsr˜   r-  Úget_cache_dirÚpathlibÚPathr¬   Úwith_optional_cacheÚapi_urlÚwith_predictionsÚTARGET_TÚwith_evaluatorsÚwith_summary_evaluatorsr*   )r/   r0   r1   r2   r3   r4   r5   r6   r7   r8   r9   r:   r;   rU   ry   Úexperiment_ÚmanagerÚ	cache_dirÚ
cache_pathrÊ   s                       r=   ri   ri   @  s^  € ð$ ×-”r×+Ò+Ó-€FÜ ×'Ñ'‰4¬T´(¼7¿;¹;Ñ2GÈÓ-P€DÜ+¨J¸fÓEÑ€Kä ØØØØ×3Ð"3ØØ'Ü.¨zÓ:àä0°ÓDØ%Ø%ñ÷ �eƒgð ô ×*Ò*¨4Ó0Ð0€yÕ0Ü—\’\ )Ó,°'×2DÑ2DÐ1EÀUÐ/KÑK‰
àˆ
Ü	×	%Ò	% jÀÇÁÐ?OÓ	PÜ˜×Ñà×.Ñ.Ü”X˜vÓ&¸ð /ð ˆGö à×-Ñ-Øð .ð ˆGö à×5Ñ5Ð6HÓIˆGä# G°hÑ?ˆØ÷! 
Q×	P×	Pús   ÄA(E4Å4
Fc                óR   •  [         R                  " U 5        g! [         a     gf = f©NTF©rP   rQ   r[   ©rS  s    r=   Ú_is_uuidry  |  ó(   € ðÜ�	Š	�%ÔØøÜó Ùðúó   ‚ ™
&¥&c                óÞ   • [        U [        R                  5      (       a  U $ [        U [        R                  5      (       d  [        U 5      (       a  UR                  U S9$ UR                  U S9$ )N©Ú
project_id)r  )rN   r   rR   rP   rQ   ry  Úread_project)rx   r8   s     r=   rt   rt   „  s_   € ô �'œ7×0Ñ0×1Ñ1ØˆÜ	�GœTŸY™Y×	'Ñ	'¬8°G×+<Ñ+<Ø×"Ñ"¨gÐ"Ð6Ð6à×"Ñ"°Ð"Ð8Ð8r?   c                óp  • U(       a  SOSn[         R                  " UR                  R                  5      nU[         R                  R
                  :X  a@  [        U [        R                  5      (       d  [        X5      n [         R                  " XUS9nOÁ[        U [        R                  5      (       a.  [        5          UR                  U R                  US9nSSS5        Ot[        U [        R                  5      (       d  [!        U 5      (       a#  [        5          UR                  XS9nSSS5        O"[        5          UR                  XS9nSSS5        U(       d  [#        W5      $ [$        R&                  " ["        5      n/ n0 nW HM  n	U	R(                  b  XiR(                     R+                  U	5        OUR+                  U	5        X˜U	R                  '   MO     UR-                  5        H  u  p«[/        US S9XŠ   l        M     U$ ! , (       d  f       N¾= f! , (       d  f       NÏ= f! , (       d  f       Nà= f)z'Load nested traces for a given project.NT)Úis_root)r~  r�  )r  r�  c                ó   • U R                   $ r,   )Údotted_order)Úrs    r=   Ú<lambda>Ú-_load_traces_for_experiment.<locals>.<lambda>¸  s   € ÀqÇ~Â~r?   )r  )r   Úget_query_backendÚinfoÚinstance_flagsÚQueryBackendÚSMITHDB_ONLYrN   r   rR   rt   Ú_load_traces_v2r   Ú	list_runsr\   rP   rQ   ry  ra   r+  r,  Úparent_run_idrÄ   rZ   ÚsortedÚ
child_runs)rx   r8   rq   r�  Úbackendry   ÚtreemaprÊ   Úall_runsr{   r  r�  s               r=   ru   ru   �  sÀ  € ö "‰d t€Gô "×3Ò3°F·K±K×4NÑ4NÓO€GØÔ%×2Ñ2×?Ñ?Ó?Ü˜'¤7×#8Ñ#8×9Ñ9Ü& wÓ7ˆGÜ"×2Ò2°7ÈGÑT‰ô 
�GœW×2Ñ2×	3Ñ	3Ü)Õ+Ø×#Ñ#¨w¯z©zÀ7Ð#ÐKˆD÷ ,Ð+ä	�GœTŸY™Y×	'Ñ	'¬8°G×+<Ñ+<Ü)Õ+Ø×#Ñ#¨wÐ#ÐHˆD÷ ,Ð+ô *Õ+Ø×#Ñ#°Ð#ÐJˆD÷ ,æÜ�D‹zÐô 	×Ò¤Ó%ð ð €GØ€HÛˆØ×ÑÑ(Ø×%Ñ%Ñ&×-Ñ-¨cÕ2à�N‰N˜3ÔØ�—‘Óñ ð &Ÿm™mžoÑˆÜ&,¨ZÑ=UÑ&VˆÑÖ#ñ .à€N÷1 ,Õ+ú÷ ,Õ+ú÷ ,Õ+ús$   Â=HÄHÄ=H'È
HÈ
H$È'
H5c                ó¨   • U R                  UR                  UR                  R                  S5      S9 Vs0 sH  nUR                  U_M     sn$ s  snf )Nrý   )r¬   rþ   )r*  r"  r3   r»   r\   )r8   rx   rÍ   s      r=   rv   rv   ¼  sc   € ð
 ×%Ñ%Ø×3Ñ3Ø×"Ñ"×&Ñ&Ð'8Ó9ð &ñ 
óñ
ˆAð 	
�‰ˆaŠñ
ñð ùò s   ¶AÚITc                 ó:   •  SSK Jn   U $ ! [         a    S s $ f = f)Nr   ©rÉ   c                ó   • U $ r,   r-   )Úxs    r=   r…  Ú_load_tqdm.<locals>.<lambda>Ï  s   € ™r?   )Ú	tqdm.autorÉ   ÚImportErrorr—  s    r=   rÁ   rÁ   Ë  s(   € ðÝ"ð €Køô ó ÙÒðús   ‚
 Š™ÚETÚ_ExperimentManagerMixin)Úboundc                  ó�   • \ rS rSr   S       SS jjr\SS j5       rSS jrS r      SS jr	SS jr
      SS	 jrS
rg)rž  iÖ  Nc               ó  • U=(       d    [         R                  " 5       U l        S U l        Uc  [	        5       U l        Oq[        U[        5      (       a7  US-   [        [        R                  " 5       R                  S S 5      -   U l        O%[        [        UR                  5      U l        Xl        U=(       d    0 nUR                  S5      (       d(  S[        R                  " 5       R                  S5      0UEnU=(       d    0 U l        X@l        g )Nrø   rú   Úrevision_id)rr   rs   r8   Ú_experimentÚ_get_random_nameÚ_experiment_namerN   rO   rP   r  r$  r   r#  r»   Úls_envÚget_langchain_env_var_metadataÚ	_metadataÚ_description)r™   r:   r3   r8   r5   s        r=   r›   Ú _ExperimentManagerMixin.__init__×  sÚ   € ð ×6¤× 4Ò 4Ó 6ˆŒØ<@ˆÔØÑÜ$4Ó$6ˆDÕ!Ü˜
¤C×(Ñ(Ø$.°Ñ$4´s¼4¿:º:»<×;KÑ;KÈBÈQÐ;OÓ7PÑ$PˆDÕ!ä$(¬¨j¯o©oÓ$>ˆDÔ!Ø)Ôà—>˜rˆØ�|‰|˜M×*Ñ*àœv×DÒDÓF×JÑJØ!ó ðð ð	ˆHð "Ÿ RˆŒØ'Õr?   c                óJ   • U R                   b  U R                   $ [        S5      e)Nz=Experiment name not provided, and experiment not yet started.)r¥  r[   rŸ   s    r=   rž   Ú'_ExperimentManagerMixin.experiment_nameô  s*   € à× Ñ Ñ,Ø×(Ñ(Ð(ÜØKó
ð 	
r?   c                óJ   • U R                   c  [        S5      eU R                   $ )NúExperiment not started yet.)r£  r[   rŸ   s    r=   r¢   Ú'_ExperimentManagerMixin._get_experimentü  s&   € Ø×ÑÑ#ÜÐ:Ó;Ð;Ø×ÑÐr?   c                óØ   • U R                   =(       d    0 nSUS'   [        R                  " 5       nU(       a  0 UESU0EnU R                  (       a  0 U R                  R                  EUEnU$ )NÚpy_sdk_evaluateÚ__ls_runnerÚgit)r¨  r¦  Úget_git_infor£  r3   )r™   Úproject_metadataÚgit_infos      r=   Ú_get_experiment_metadataÚ0_ExperimentManagerMixin._get_experiment_metadata  s~   € ØŸ>™>×/¨RÐØ*;Ð˜Ñ'Ü×&Ò&Ó(ˆÞð Ø"ð à�xñ Ðð ××ð Ø×"Ñ"×+Ñ+ð à"ð Ðð  Ðr?   c                ó¶  • U R                   nSn[        U SS 5      n[        U SS 5      n[        U SS 5      n[        U5       H7  n U R                  R	                  U R                   U R
                  UUUUUS9s  $    [        SU S	35      e! [        R                   a9    U S[        [        R                  " 5       R                  S S 5       3U l          M–  f = f)
Né
   Ú_num_examplesÚ_num_repetitionsÚ_evaluator_keys)r5   r"  r3   Únum_examplesr7   r`  rø   é   z+Could not find a unique experiment name in z= attempts. Please try again with a different experiment name.)r¥  Úgetattrr)  r8   Úcreate_projectr©  r-  ÚLangSmithConflictErrorrO   rP   r  r$  r[   )	r™   r¬   r3   Ústarting_nameÚnum_attemptsr¾  r7   r`  rC  s	            r=   Ú_create_experimentÚ*_ExperimentManagerMixin._create_experiment  sü   € ð ×-Ñ-ˆØˆÜ˜t _°dÓ;ˆÜ! $Ð(:¸DÓAˆÜ  Ð'8¸$Ó?ˆÜ�|Ö$ˆAðWØ—{‘{×1Ñ1Ø×)Ñ)Ø $× 1Ñ 1Ø)3Ø%Ø!-Ø$3Ø#1ð 2ð ò ñ %ô Ø9¸,¸ð HBð Bó
ð 	
øô ×2Ñ2ó WØ+8¨/¸¼3¼t¿zºz»|×?OÑ?OÐPRÐQRÐ?SÓ;TÐ:UÐ(V�×%ðWús   Á2BÂA	CÃCc                ó”   • U R                   c.  U R                  5       nU R                  UR                  U5      nU$ U R                   nU$ r,   )r£  r·  rÅ  r¬   )r™   Úfirst_examplerµ  rx   s       r=   Ú_get_projectÚ$_ExperimentManagerMixin._get_project,  sR   € Ø×ÑÑ#Ø#×<Ñ<Ó>ÐØ×-Ñ-Ø×(Ñ(Ð*:óˆGð
 ˆð ×&Ñ&ˆGØˆr?   c                ó>  • U(       a€  UR                   (       ao  UR                   R                  S5      S   nUR                  nUR                  S5      S   nU SU SUR                   3n[	        SU R
                   SU S35        g [	        S	U R
                  5        g )
Nr¦   r   r§   r¨   r©   z-View the evaluation results for experiment: 'z' at:
rZ  z%Starting evaluation of experiment: %s)rª   r«   r¬   r\   r[  rž   )r™   rx   rÈ  r­   r¬   r®   r´   s          r=   Ú_print_experiment_startÚ/_ExperimentManagerMixin._print_experiment_start6  s©   € ö �w—{—{Ø!Ÿ+™+×+Ñ+¨CÓ0°Ñ3ˆKØ&×1Ñ1ˆJØ"×(Ñ(¨Ó8¸Ñ;ˆHà�*˜J z lð 3$Ø$+§J¡J <ð1ð ô Ø?À×@TÑ@TÐ?Uð VØ'Ð(¨ð.õô Ø7¸×9MÑ9Mõr?   )r©  r£  r¥  r¨  r8   rU  )r:   ú+Optional[Union[schemas.TracerSession, str]]r3   úOptional[dict]r8   úOptional[langsmith.Client]r5   rî   rê   )rë   úschemas.TracerSession)r¬   rì   r3   rV  rë   rÑ  )rÈ  r   rë   rÑ  )rx   zOptional[schemas.TracerSession]rÈ  r   rë   rð   )rƒ   r„   r…   r†   r›   rö   rž   r¢   r·  rÅ  rÉ  rÌ  rˆ   r-   r?   r=   rž  rž  Ö  s˜   † ð
 $(Ø-1Ø%)ð(ð @ð(ð !ð	(ð
 +ð(ð #õ(ð: ó
ó ð
ô ò
 ð 
Ø#ð
Ø/3ð
à	ô
ô6ðØ6ðØGVðà	÷r?   c                  óä  ^ • \ rS rSrSr              S                               SU 4S jjjr    SS jr\SS j5       r\SS j5       r	\SS j5       r
\S S	 j5       rS!S
 jr S"     S#S jjrSS.     S$S jjr    S%S jrS&S jrS'S jr  S(       S)S jjr        S*S jr S"     S+S jjr    S,S jrS-S jrS.S jrS/S jrS0S jrSrU =r$ )1ré   iL  a  Manage the execution of experiments.

Supports lazily running predictions and evaluations in parallel to facilitate
result streaming and early debugging.

Args:
    data (DATA_T): The data used for the experiment. Can be a dataset name or ID OR
        a generator of examples.
    num_repetitions (int): The number of times to run over the data.
    runs (Optional[Iterable[schemas.Run]]): The runs associated with the experiment
        predictions.
    experiment (Optional[schemas.TracerSession]): The tracer session
        associated with the experiment.
    experiment_prefix (Optional[str]): The prefix for the experiment name.
    metadata (Optional[dict]): Additional metadata for the experiment.
    client (Optional[langsmith.Client]): The Langsmith client used for
        the experiment.
    evaluation_results (Optional[Iterable[EvaluationResults]]): The evaluation
        sresults for the experiment.
    summary_results (Optional[Iterable[EvaluationResults]]): The aggregate results
        for the experiment.
Nc               óô   >• [         TU ]  UUUUS9  Xl        X°l        S U l        XPl        X`l        Xpl        X�l        U
b  U
O[        XR                  S9U l        XÀl        XÐl        Xàl        Xðl        UU l        g )N)r:   r3   r8   r5   )r8   )Úsuperr›   Ú_datar½  rF  Ú_runsÚ_evaluation_resultsrÆ   r¼  Ú_resolve_num_examplesr8   r»  rg  Ú_reuse_attachmentsÚ_upload_resultsÚ_attachment_raw_data_dictÚ_error_handling)r™   r0   r:   r3   r8   ry   r�   Úsummary_resultsr5   r7   r¾  r`  ra  Úreuse_attachmentsr;   Úattachment_raw_data_dictrU   Ú	__class__s                    €r=   r›   Ú_ExperimentManager.__init__d  s‘   ø€ ô( 	‰ÑØ!ØØØ#ð	 	ñ 	
ð Œ
Ø-ÔØ>BˆŒØŒ
Ø#5Ô Ø /ÔØ /Ôð Ñ'ñ ä& t·K±KÑ@ð 	Ôð
 %8Ô!Ø"3ÔØ-ÔØ)AÔ&Ø-ˆÕr?   c                ó”  • [        US5      (       a  UR                  (       d  U$ 0 nUR                  R                  5        Hƒ  u  p4U R                  bm  [	        UR
                  5      U-   U R                  ;   aG  US   [        R                  " U R                  [	        UR
                  5      U-      5      US   S.X#'   M  XBU'   M…     [        R                  " UR
                  UR                  UR                  UR                  UR                  UR                  UR                  UR                   UUR"                  UR$                  S9$ )a¡  Reset attachment readers for an example.

This is only in the case that an attachment is going to be used by more
than 1 callable (target + evaluators). In that case we keep a single copy
of the attachment data in `self._attachment_raw_data_dict`, and create
readers from that data. This makes it so that we don't have to keep
copies of the same data in memory, instead we can just create readers
from the same data.
ÚattachmentsÚpresigned_urlÚ	mime_type)rä  Úreaderrå  )r\   Ú
created_atr¬   ÚinputsÚoutputsr3   Úmodified_atr  rã  Ú	_host_urlÚ
_tenant_id)Úhasattrrã  rZ   rÛ  rO   r\   ÚioÚBytesIOr   ÚExamplerç  r¬   rè  ré  r3   rê  r  rë  rì  )r™   r€   Únew_attachmentsr#  Ú
attachments        r=   Ú!_reset_example_attachment_readersÚ4_ExperimentManager._reset_example_attachment_readers�  s  € ô �w ×.Ñ.°g×6I×6IØˆNà=?ˆØ '× 3Ñ 3× 9Ñ 9Ö ;ÑˆDà×.Ñ.Ñ:Ü˜Ÿ
™
“O dÑ*¨d×.LÑ.LÓLð &0°Ñ%@Ü ŸjšjØ×6Ñ6´s¸7¿:¹:³ÈÑ7MÑNóð ",¨KÑ!8ñ)�Ó%ð )3 Ó%ñ !<ô  �ŠØ�z‰zØ×)Ñ)Ø×)Ñ)Ø—>‘>Ø—O‘OØ×%Ñ%Ø×+Ñ+Ø!×/Ñ/Ø'Ø×'Ñ'Ø×)Ñ)ñ
ð 	
r?   c           	     ó  ^ ^• T R                   GcJ  [        T R                  T R                  T R                  S9T l         T R
                  (       a¤  T R                  c—  [        R                  " T R                   5      u  nT l         U VVVs0 sHY  nUR                  =(       d    0 R                  5        H/  u  p4[        UR                  5      U-   US   R                  5       _M1     M[     snnnT l        T R                  S:”  aW  [        T R                   5      m[        R                   R#                  UU 4S j[%        T R                  5       5       5      T l         [        R                  " T R                   5      u  T l         nU$ s  snnnf )N)r8   ra  ræ  r)   c              3  ón   >#   • U H&  nT Vs/ sH  nTR                  U5      PM     snv •  M(     g s  snf 7fr,   )ró  )rE   rC  r€   Úexamples_listr™   s      €€r=   rH   Ú._ExperimentManager.examples.<locals>.<genexpr>Ï  sD   øé € ð ?ñ
 :˜ñ (5óá'4˜Gð ×>Ñ>¸wÖGÙ'4öò :ùò	ùs   ƒ	5Œ0¦5)rF  Ú_resolve_datarÕ  r8   rg  rÙ  rÛ  Ú	itertoolsÚteerã  rZ   rO   r\   Úreadr¼  ra   ÚchainÚfrom_iterabler)  )r™   Úexamples_copyrÍ   r#  rS  Úexamples_iterr÷  s   `     @r=   rI  Ú_ExperimentManager.examples¾  s5  ù€ à�>‰>Ò!Ü*Ø—
‘
Ø—{‘{Ø$(×$=Ñ$=ñˆDŒNð
 ×&×&¨4×+IÑ+IÑ+QÜ09·²¸d¿n¹nÓ0MÑ-�˜tœ~ñ +õ2á*˜Ø()¯©×(;¸×'BÑ'BÖ'D™˜ô ˜Ÿ™“I Ñ$ e¨H¡o×&:Ñ&:Ó&<Ò<á'Dñ %Ù*ó2�Ô.ð
 ×$Ñ$ qÓ(Ü $ T§^¡^Ó 4�Ü!*§¡×!>Ñ!>õ ?ô
 # 4×#8Ñ#8Ô9ó?ó "�”ô )2¯ª°d·n±nÓ(EÑ%ˆŒ˜ØÐùô2s   ÂAFc                ó(  • U R                   b  [        U R                   SS 5      (       d3  [        [        U R                  5      5      n[        UR                  5      $ [        [        [        R                  U R                   5      R                  5      $ )Nr"  )r£  rÀ  ÚnextÚiterrI  rO   r¬   r   r   r&  r"  )r™   r€   s     r=   r¬   Ú_ExperimentManager.dataset_idÙ  sw   € à×ÑÑ#¬7Ø×ÑÐ4°d÷,
ñ ,
ô œ4 §¡Ó.Ó/ˆGÜ�w×)Ñ)Ó*Ð*ÜÜ”×,Ñ,¨d×.>Ñ.>Ó?×TÑTó
ð 	
r?   c                óZ   • U R                   c  S U R                   5       $ U R                   $ )Nc              3  ó(   #   • U H	  nS / 0v •  M     g7f)rÊ   Nr-   )rE   rC  s     r=   rH   Ú8_ExperimentManager.evaluation_results.<locals>.<genexpr>ç  s   é € Ð;©]¨�Y •Oª]ùó   ‚)r×  rI  rŸ   s    r=   r�   Ú%_ExperimentManager.evaluation_resultsä  s)   € à×#Ñ#Ñ+Ù;¨T¯]ª]Ó;Ð;Ø×'Ñ'Ð'r?   c                ó†   • U R                   c  [        S5      e[        R                  " U R                   5      u  U l         nU$ )Nz;Runs not provided in this experiment. Please predict first.)rÖ  r[   rú  rû  )r™   Ú	runs_iters     r=   ry   Ú_ExperimentManager.runsê  s=   € à�:‰:ÑÜØMóð ô !*§¢¨d¯j©jÓ 9ÑˆŒ
�IØÐr?   c                ó&  • [        [        R                  " U R                  S5      5      nU R                  (       a  U R                  U5      OS nU R                  X!5        U R                  U R                  S'   U R                  U R                  US9$ )Nr)   r7   )r:   )
r  rú  ÚislicerI  rÚ  rÉ  rÌ  r¼  r¨  Ú_copy)r™   rÈ  rx   s      r=   r˜   Ú_ExperimentManager.startó  sr   € ÜœY×-Ò-¨d¯m©m¸QÓ?Ó@ˆØ6:×6J×6J�$×#Ñ# MÔ2ÐPTˆØ×$Ñ$ WÔ<Ø,0×,AÑ,Aˆ�‰Ð(Ñ)Ø�z‰z˜$Ÿ-™-°GˆzÐ<Ð<r?   c               óÎ   • [        5       nUR                  U R                  UU[        U5      S9n[        R
                  " US5      u  pVU R                  S U 5       S U 5       S9$ )z3Lazily apply the target function to the experiment.)r6   ra  rL   c              3  ó(   #   • U H	  oS    v •  M     g7f©r€   Nr-   ©rE   Úpreds     r=   rH   Ú6_ExperimentManager.with_predictions.<locals>.<genexpr>
  s   é € Ð,© �)Ž_ªùr	  c              3  ó(   #   • U H	  oS    v •  M     g7f©r{   Nr-   r  s     r=   rH   r  
  s   é € Ð3OÉBÀD¸¶KÊBùr	  )ry   )r   r{   Ú_predictÚ_target_include_attachmentsrú  rû  r  )r™   r/   r6   ÚcontextÚ_experiment_resultsÚr1Úr2s          r=   rm  Ú#_ExperimentManager.with_predictionsú  so   € ô “.ˆØ%Ÿk™kØ�M‰MØØ+Ü ;¸FÓ Cð	 *ð 
Ðô —’Ð2°AÓ6‰ˆØ�z‰zÙ,©Ó,Ñ3OÉBÓ3Oð ð 
ð 	
r?   rc  c               óà   • [        U5      n[        5       nUR                  U R                  XS9n[        R
                  " US5      u  pVnU R                  S U 5       S U 5       S U 5       S9$ )z7Lazily apply the provided evaluators to the experiment.rc  é   c              3  ó(   #   • U H	  oS    v •  M     g7fr  r-   ©rE   r  s     r=   rH   Ú5_ExperimentManager.with_evaluators.<locals>.<genexpr>"  s   é € Ð0©R 6�IÖªRùr	  c              3  ó(   #   • U H	  oS    v •  M     g7fr  r-   r$  s     r=   rH   r%  #  s   é € Ð1©b F˜–-ªbùr	  c              3  ó(   #   • U H	  oS    v •  M     g7f)r�   Nr-   r$  s     r=   rH   r%  $  s   é € ÐNÉ2ÀÐ';Ö <Ê2ùr	  )ry   r�   )Ú_resolve_evaluatorsr   r{   Ú_scorerú  rû  r  )r™   r1   r6   r  Úexperiment_resultsr  r  Úr3s           r=   ro  Ú"_ExperimentManager.with_evaluators  sw   € ô )¨Ó4ˆ
Ü“.ˆØ$Ÿ[™[Ø�K‰K˜ð )ð 
Ðô
 —]’]Ð#5°qÓ9‰
ˆ�Ø�z‰zÙ0©RÓ0Ù1©bÓ1ÙNÉ2ÓNð ð 
ð 	
r?   c                ó®   • [        U5      n[        5       nUR                  U R                  U5      nU R	                  U R
                  U R                  US9$ )z?Lazily apply the provided summary evaluators to the experiment.)ry   rÝ  )Ú_wrap_summary_evaluatorsr   r{   Ú_apply_summary_evaluatorsr  rI  ry   )r™   r2   Úwrapped_evaluatorsr  Úaggregate_feedback_gens        r=   rp  Ú*_ExperimentManager.with_summary_evaluators'  sZ   € ô
 6Ð6HÓIÐÜ“.ˆØ!(§¡Ø×*Ñ*Ð,>ó"
Ðð �z‰zØ�M‰M §	¡	Ð;Qð ð 
ð 	
r?   c              #  ó�   #   • [        U R                  U R                  U R                  5       H  u  pn[	        UUUS9v •  M     g7f)z?Return the traces, evaluation results, and associated examples.©r{   r€   r�   N)Úzipry   rI  r�   r}   )r™   r{   r€   r�   s       r=   rÂ   Ú_ExperimentManager.get_results5  sG   é € ä03Ø�I‰I�t—}‘} d×&=Ñ&=ö1
Ñ,ˆCÐ,ô &ØØØ#5ñô ò1
ùs   ‚AAc                óˆ   • U R                   c  S/ 0$ SU R                    VVs/ sH  nUS    H  nUPM     M     snn0$ s  snnf )zEIf `summary_evaluators` were applied, consume and return the results.rÊ   )rÆ   )r™   rÊ   Úress      r=   rÅ   Ú%_ExperimentManager.get_summary_scores@  s_   € à× Ñ Ñ(Ø˜r�?Ð"ð à#×4Ò4ôá4�GØ" 9Ô-�Có á-ñ Ù4òð
ð 	
ùós   ¢>c             #  óš  #   • [        U5      nUS:X  aZ  U R                   HI  n[        UUU R                  U R                  U R
                  U R                  UU R                  5      v •  MK     O¶[        R                  " U5       nU R                   Vs/ sHR  nUR                  [        UUU R                  U R                  U R
                  U R                  UU R                  5	      PMT     nn[        R                  " U5       H  nUR                  5       v •  M     SSS5        U R                  5         gs  snf ! , (       d  f       N$= f7f)z(Run the target function on the examples.r   N)Ú_ensure_traceablerI  Ú_forwardrž   r¨  r8   rÚ  rÜ  r-  r.  r  r/  Úas_completedr  Ú_end)	r™   r/   r6   ra  Úfnr€   r  r@  rB  s	            r=   r  Ú_ExperimentManager._predictO  s   é € ô ˜vÓ&ˆà˜aÓØŸ=œ=�ÜØØØ×(Ñ(Ø—N‘NØ—K‘KØ×(Ñ(Ø'Ø×(Ñ(ó	ô 	ò )ô ×3Ò3°OÔDÈð $(§=¢=óñ $1˜ð —O‘OÜ ØØØ×,Ñ,ØŸ™ØŸ™Ø×,Ñ,Ø+Ø×,Ñ,ö
ñ $1ð ð ô !Ÿošo¨gÖ6�FØ Ÿ-™-›/Ô)ñ 7÷ Eð$ 	�	‰	�ùò#÷ EÕDüs1   ‚BEÂD:ÂAD5Ã+1D:ÄEÄ5D:Ä:
EÅEc                óÌ  • [         R                  " 5       n0 US   =(       d    0 EU R                  US   R                  US   R                  S.En[         R                  " S0 0 UESUU R
                  (       d  SOSU R                  S.ED6   US   nUS   nUS	   nU Há  n	[        R                  " 5       n
 U	R                  UUU
S
9nUS   R                  U R                  R                  U5      5        U R
                  (       a4  U R                  R                  UUU R                  5       R                  US9  UR,                  c  M«  UR,                   H&  nUR,                  U   S   nUR/                  S5        M(     Mã     [1        UUUS9sS S S 5        $ ! [         Ga0  n U	R                  n[!        U Vs/ sH  n[#        UU
[%        U5      SS0S9PM     Os  snf snS9nUS   R                  U R                  R                  U5      5        U R
                  (       a4  U R                  R                  UUU R                  5       R                  US9  O/! [         a"  n[&        R)                  SU 35         S nAOS nAff = f[&        R+                  S[%        U	5       SU(       a  UR                  OS S[%        U5       3SS9   S nAGN“S nAff = f! , (       d  f       g = f)Nr3   r€   r{   )r:   rw   Úreference_run_idr1   ÚlocalT)r  r3   Úenabledr8   r�   )r{   r€   Úevaluator_run_idrÊ   )r{   r~  Ú	_executorÚerror)r  r  r  Úextra)rÊ   zError parsing feedback keys: zError running evaluator z on run Ú ú: ©Úexc_inforæ  r   r4  r-   )rg   Úget_tracing_contextrž   r\   r  rÚ  r8   rP   r  Úevaluate_runÚextendÚ_select_eval_resultsÚ_log_evaluation_feedbackr¢   Ú	ExceptionÚfeedback_keysr"   r!   Úreprr]   r^   rG  rã  Úseekr}   )r™   r1   Úcurrent_resultsr  Úcurrent_contextr3   r{   r€   Úeval_resultsr>  rE  Úevaluator_responserÍ   rS  r  Úerror_responseÚe2rò  ræ  s                      r=   Ú_run_evaluatorsÚ"_ExperimentManager._run_evaluators{  s  € ô ×0Ò0Ó2ˆð
Ø˜zÑ*×0¨bð
ð #×2Ñ2Ø(7¸	Ñ(B×(EÑ(EØ$3°EÑ$:×$=Ñ$=ñð
ˆô ×Òñ 
ðØ!ðà ,Ø$Ø*.×*>×*>™7ÀDØŸ+™+òó
ð " %Ñ(ˆCØ% iÑ0ˆGØ*Ð+?Ñ@ˆLÛ'�	Ü#'§:¢:£<Ð ð3Ø)2×)?Ñ)?ØØ 'Ø)9ð *@ð *Ð&ð ! Ñ+×2Ñ2ØŸ™×8Ñ8Ð9KÓLôð ×+×+àŸ™×<Ñ<Ø.Ø #Ø'+×';Ñ';Ó'=×'@Ñ'@Ø&.ð	 =ñ ðP ×&Ñ&Ó2Ø&-×&9Ô&9˜
Ø!(×!4Ñ!4°ZÑ!@ÀÑ!J˜ØŸ™ Ažó ':ño (ôv 'ØØØ#/ñ÷O
ñ 
øô@ !ô !ðØ(1×(?Ñ(?˜ä):ñ ,9ó%ñ ,9 Cô !1Ø(+Ø2BÜ,0°«GØ+2°D¨/ô	!"ò ,9ùô%ñ
*˜ð % YÑ/×6Ñ6Ø ŸK™K×<Ñ<¸^ÓLôð  ×/×/à ŸK™K×@Ñ@Ø .Ø$'Ø+/×+?Ñ+?Ó+A×+DÑ+DØ*2ð	 Añ ùô %ó ÜŸ™Ð'DÀRÀDÐ%IÔJÜûðúô —L‘LØ2´4¸	³?Ð2Cð D Þ*- §¢°2Ð6°b¼¸a»¸	ðCà!%ð !÷ ûð;!ú÷A
õ 
úsp   Â*KÂ7BFÄ:KÅ	AKÆKÆ#IÆ8 GÇA=IÉKÉ
J	É I=	É8KÉ=J	ÊAKËKËKËKË
K#c           
   #  óÊ  #   • [         R                  " U=(       d    SS9 nUS:X  aB  [        5       nU R                  5        H#  nUR	                  U R
                  UUU5      v •  M%     O¿[        5       nU R                  5        Hp  nUR                  UR                  U R
                  UUU5      5         [        R                  " USS9 H&  nUR                  5       v •  UR                  U5        M(     Mr     [        R                  " U5       H  nUR                  5       nUv •  M     SSS5        g! [        R                  [        4 a     MË  f = f! , (       d  f       g= f7f)z‚Run the evaluators on the prediction stream.

Expects runs to be available in the manager.
(e.g. from a previous prediction step)
r)   r  r   gü©ñÒMbP?)r¸   N)r-  r.  r   rÂ   r{   r\  rÈ   Úaddr  r/  r=  r  ÚremoveÚTimeoutError)	r™   r1   r6   r  r  rV  r@  rB  r  s	            r=   r)  Ú_ExperimentManager._score×  s6  é € ô ×/Ò/Ø'×,¨1ò
àØ !Ó#Ü&›.�Ø'+×'7Ñ'7Ö'9�OØ!Ÿ+™+Ø×,Ñ,Ø"Ø'Ø ó	ô ò (:ô ›%�Ø'+×'7Ñ'7Ö'9�OØ—K‘KØ Ÿ™Ø ×0Ñ0Ø&Ø+Ø$ó	ôðô ')§o¢o°gÀuÔ&M˜FØ"(§-¡-£/Ò1Ø#ŸN™N¨6Ö2ó 'Nñ (:ô" !Ÿošo¨gÖ6�FØ#Ÿ]™]›_�FØ ”Lñ 7÷?
ð 
øô: ŸO™O¬\Ð:ó Úðú÷;
õ 
üsA   ‚E# BEÂ4?D1Ã35EÄ(	E#Ä1EÅEÅEÅEÅ
E ÅE#c              #  ó^  #   • / / p2[        U R                  U R                  5       H'  u  pEUR                  U5        UR                  U5        M)     / n[        R
                  " 5        nU R                  (       a  U R                  5       R                  OS n[        R                  " 5       n	0 U	S   =(       d    0 EU R                  US.En
[        R                  " S0 0 U	ESU
U R                  U R                  (       d  SOSS.ED6   U H¬  n U" X#5      nU R                  R                  UUR                  S9nUR!                  U5        U R                  (       aZ  U HR  nUR#                  S1S	9nUR%                  S
S 5      nUR&                  " U R                  R(                  40 UDS UUS.D6  MT     M¬  M®     S S S 5        S S S 5        SU0v •  g ! [*         a.  n[,        R/                  S[1        U5       SU 3SS9   S nAMú  S nAff = f! , (       d  f       NX= f! , (       d  f       Na= f7f)Nr3   )r:   r£   r1   rC  T)r  r3   r8   rD  )Úfn_nameÚtarget_run_id)ÚexcludeÚevaluator_info)r  r~  Úsource_infoz Error running summary evaluator rJ  rK  rÊ   r-   )r5  ry   rI  rÄ   r-  r.  rÚ  r¢   r\   rg   rM  rž   r  r8   rP  rƒ   rO  Ú
model_dumpÚpopr  r  rR  r]   rG  rT  )r™   r2   ry   rI  r{   r€   Úaggregate_feedbackr  r~  rW  r3   r>  Úsummary_eval_resultÚflattened_resultsr  Úfeedbackrg  rÍ   s                     r=   r/  Ú,_ExperimentManager._apply_summary_evaluators  s$  é € ð ˜RˆhÜ §	¡	¨4¯=©=Ö9‰LˆCØ�K‰K˜ÔØ�O‰O˜GÖ$ñ :ð  ÐÜ×/Ò/Ô1°XØ6:×6J×6J˜×-Ñ-Ó/×2Ò2ÐPTˆJÜ ×4Ò4Ó6ˆOðØ" :Ñ.×4°"ðð #'×"6Ñ"6Ø%/ñðˆHô ×#Ò#ñ ðØ%ðà$0Ø (Ø"Ÿk™kØ.2×.B×.B™wÈòóó "4�IðÙ.7¸Ó.GÐ+à,0¯K©K×,LÑ,LØ/Ø$-×$6Ñ$6ð -Mð -Ð)ð +×1Ñ1Ð2CÔDØ×/×/Û*; Ø+1×+<Ñ+<ÀoÐEVÐ+<Ð+W Ø19·±Ð>NÐPTÓ1U Ø (§¢Ø$(§K¡K×$?Ñ$?ñ!"à&.ð!"ð ,0Ø/9Ø0>ö!"ó +<ñ 0ñ "4÷÷ 2ðX Ð,Ð-Ó-øô %ó ÜŸ™Ø>¼tÀI»Ð>OÈrÐRSÐQTÐUØ%)ð %÷ ûðú÷;õ ú÷ 2Õ1üsb   ‚A&H-Á(BHÄHÄB&GÆ3HÆ9HÇH-Ç
H	Ç#H	Ç=HÈH	ÈHÈ
H	ÈHÈ
H*È&H-c                óê   • [        U R                  5      nU Vs/ sH!  o"R                  (       d  M  UR                  PM#     nnU(       a  [        U5      OS nU(       a  UR	                  5       $ S $ s  snf r,   )ra   rI  rê  ÚmaxÚ	isoformat)r™   rI  Úexrê  Úmax_modified_ats        r=   Ú_get_dataset_versionÚ'_ExperimentManager._get_dataset_version:  sZ   € Ü˜Ÿ™Ó&ˆÙ08ÓK±¨"¿N½N“~�r—~”~±ˆÐKö /:œ#˜kÔ*¸tˆÞ.=ˆ×(Ñ(Ó*ÐGÀ4ÐGùò	 Ls
   šA0±A0c                ó°  • [        U R                  5      n[        5       nU H§  nUR                  (       a‚  UR                  R	                  S5      (       ab  [        UR                  S   [         5      (       a@  UR                  S    H+  n[        U[        5      (       d  M  UR                  U5        M-     M–  UR                  S5        M©     [        U5      $ )NÚdataset_splitÚbase)ra   rI  rÈ   r3   r»   rN   rO   r_  )r™   rI  Úsplitsr€   r«   s        r=   Ú_get_dataset_splitsÚ&_ExperimentManager._get_dataset_splitsB  s¡   € Ü˜Ÿ™Ó&ˆÜ“ˆÛˆGà× × Ø×$Ñ$×(Ñ(¨×9Ñ9Ü˜w×/Ñ/°Ñ@Ä$×GÑGà$×-Ñ-¨oÔ>�EÜ! %¬×-Ó-ØŸ
™
 5Ö)ó ?ð —
‘
˜6Ö"ñ  ô �F‹|Ðr?   c                ó,  • U R                   (       d  g U R                  nUc  [        S5      eU R                  5       nU R	                  5       US'   U R                  5       US'   U R                  R                  UR                  0 UR                  EUES9  g )Nr®  rý   Údataset_splits)r3   )
rÚ  r£  r[   r·  ru  r{  r8   Úupdate_projectr\   r3   )r™   r:   rµ  s      r=   r>  Ú_ExperimentManager._endS  s�   € Ø×#×#ØØ×%Ñ%ˆ
ØÑÜÐ:Ó;Ð;à×8Ñ8Ó:ÐØ.2×.GÑ.GÓ.IÐÐ*Ñ+Ø-1×-EÑ-EÓ-GÐÐ)Ñ*Ø�‰×"Ñ"Ø�M‰MðØ×%Ñ%ðà"ðð 	#ò 	
r?   c                ó¶  • U R                   4nU R                  U R                  U R                  U R                  U R
                  U R                  U R                  U R                  U R                  U R                  U R                  U R                  U R                  S.n[        U5      [        U[        U5      S  5      -   n0 UEUEnU R                   " U0 UD6$ )N)r:   r3   ry   r8   r�   rÝ  r¾  r`  ra  rÞ  r;   rß  rU   )rÕ  r£  r¨  rÖ  r8   r×  rÆ   r»  r½  rg  rÙ  rÚ  rÛ  rÜ  ra   rb   rà  )r™   Úargsr<   Údefault_argsÚdefault_kwargsÚ	full_argsÚfull_kwargss          r=   r  Ú_ExperimentManager._copye  sÆ   € ØŸ
™
�}ˆà×*Ñ*ØŸ™Ø—J‘JØ—k‘kØ"&×":Ñ":Ø#×4Ñ4Ø ×.Ñ.Ø"×2Ñ2Ø#'×#<Ñ#<Ø!%×!8Ñ!8Ø"×2Ñ2Ø(,×(FÑ(FØ"×2Ñ2ñ
ˆô ˜“J¤ l´3°t³9°;Ð&?Ó!@Ñ@ˆ	Ø2˜Ð2¨6Ð2ˆØ�~Š~˜yÐ8¨KÑ8Ð8r?   )rÛ  rÕ  rÜ  r×  r½  rF  rg  r»  r¼  rÙ  rÖ  rÆ   rÚ  )NNNNNNr)   NNFFTNÚlog) r:   rÎ  r3   rÏ  r8   rÐ  ry   úOptional[Iterable[schemas.Run]]r�   ú%Optional[Iterable[EvaluationResults]]rÝ  rŠ  r5   rî   r7   rñ   r¾  ró   r`  úOptional[list[str]]ra  rV   rÞ  rV   r;   rV   rß  rÏ  rU   úLiteral['log', 'ignore']r0   ÚDATA_T)r€   r   rë   r   )rë   úIterable[schemas.Example]rê   )rë   zIterable[EvaluationResults])rë   zIterable[schemas.Run])rë   ré   r,   )r6   ró   r/   rn  rë   ré   )r1   z*Sequence[Union[EVALUATOR_T, RunEvaluator]]r6   ró   rë   ré   )r2   úSequence[SUMMARY_EVALUATOR_T]rë   ré   )rë   úIterable[ExperimentResultRow])rë   zdict[str, list[dict]])NF)r6   ró   ra  rV   r/   rn  rë   z&Generator[_ForwardResults, None, None])r1   úSequence[RunEvaluator]rV  r}   r  zcf.ThreadPoolExecutorrë   r}   )r1   r‘  r6   ró   rë   r�  )r2   r�  rë   z(Generator[EvaluationResults, None, None]rí   )rë   r‹  rï   )r‚  r   r<   r   rë   ré   )rƒ   r„   r…   r†   rõ   r›   ró  rö   rI  r¬   r�   ry   r˜   rm  ro  rp  rÂ   rÅ   r  r\  r)  r/  ru  r{  r>  r  rˆ   Ú__classcell__)rà  s   @r=   ré   ré   L  su  ø† ñð8 $(Ø-1Ø04ØDHØAEØ%)Ø Ø&*Ø.2Ø$)Ø"'Ø#Ø37Ø38ð%*.ð @ð	*.ð
 !ð*.ð +ð*.ð .ð*.ð Bð*.ð ?ð*.ð #ð*.ð ð*.ð $ð*.ð ,ð*.ð "ð*.ð  ð*.ð  ð!*.ð" #1ð#*.ð$ 1ð%*.à÷*.ð *.ðX,
Ø&ð,
à	ô,
ð\ óó ðð4 ó
ó ð
ð ó(ó ð(ð
 óó ðô=ð *.ð	
ð 'ð	
àð
ð
 
õ
ð8 *.ñ
ð
ð
ð 'ð
ð 
õ
ð4
à9ð
ð 
ô
ô	ô
ð& *.Ø$)ð*ð 'ð	*ð
 "ð*àð*ð 
0õ*ðXZà*ðZð -ðZð (ð	Zð
 
ôZð~ *.ð+!à*ð+!ð 'ð+!ð 
'õ	+!ðZ4.Ø"?ð4.à	1ô4.ôlHôô"
÷$9ò 9r?   ré   c                ó    • / nU  HE  n[        U[        5      (       a  UR                  U5        M+  UR                  [        U5      5        MG     U$ r,   )rN   r#   rÄ   r&   )r1   rÊ   r>  s      r=   r(  r(  {  sD   € ð €GÛˆ	Ü�i¤×.Ñ.Ø�N‰N˜9Ö%à�N‰Nœ=¨Ó3Ö4ñ	  ð
 €Nr?   c                óT   • SS jn/ nU  H  nUR                  U" U5      5        M     U$ )Nc                óŠ   ^ ^• [        T SS5      m[        T 5      m [        R                  " T 5            SUU 4S jj5       nU$ )Nrƒ   ÚBatchEvaluatorc                óœ   >^ ^• [         R                  " TS9      SUUU 4S jj5       nU" S[        T 5       S3S[        T5       S35      $ )N©r#  c                ó:   >• T" [        T5      [        T5      5      $ r,   )ra   )Úruns_Ú	examples_r>  rI  ry   s     €€€r=   Ú_wrapper_super_innerÚ]_wrap_summary_evaluators.<locals>._wrap.<locals>._wrapper_inner.<locals>._wrapper_super_inner’  s   ø€ ñ !¤ d£¬T°(«^Ó<Ð<r?   zRuns[] (Length=Ú)zExamples[] (Length=)rš  rO   r›  rO   rë   ú*Union[EvaluationResult, EvaluationResults])rg   Ú	traceablerb   )ry   rI  rœ  Ú	eval_namer>  s   `` €€r=   Ú_wrapper_innerÚ?_wrap_summary_evaluators.<locals>._wrap.<locals>._wrapper_innerŽ  sm   ú€ ô �\Š\˜yÑ)ð=Øð=Ø'*ð=à;÷=ð =ó *ð=ñ
 (Ø!¤# d£) ¨AÐ.Ð2EÄcÈ(ÃmÀ_ÐTUÐ0Vóð r?   )ry   zSequence[schemas.Run]rI  zSequence[schemas.Example]rë   rŸ  )rÀ  r$   Ú	functoolsÚwraps)r>  r¢  r¡  s   ` @r=   Ú_wrapÚ'_wrap_summary_evaluators.<locals>._wrapŠ  sW   ù€ Ü˜I zÐ3CÓDˆ	Ü0°Ó;ˆ	ä	�Š˜Ó	#ð	Ø'ð	Ø3Lð	à7÷	ó 
$ð	ð Ðr?   )r>  r   rë   r   )rÄ   )r1   r¦  rÊ   r>  s       r=   r.  r.  ‡  s.   € ôð( €GÛˆ	Ø�‰‘u˜YÓ'Ö(ñ  à€Nr?   c                  ó*   • \ rS rSr% S\S'   S\S'   Srg)Ú_ForwardResultsi¤  r~   r{   r   r€   r-   Nr‚   r-   r?   r=   r©  r©  ¤  s   ‡ Ø	ÓØÖr?   r©  c                ó,  ^^• S mSU4S jjnSU4S jjn	TR                   =(       d    TR                  R                  5       n
[        R                  " UU0 UESU
0EUS9nUS:X  a  TR
                  US'   OUS:X  a  X›S'   O[        S	U< 35      e[        R                  " U(       d  S
OSS9    [        U 5      nU Vs/ sH  n[        TU5      PM     nnU " USU06  U(       aC  TR                  b6  TR                   H&  nTR                  U   S   nUR                  S5        M(     [        [!        ["        R$                  T5      TS9sS S S 5        $ s  snf ! [         a"  n[        R                  SU 3SSS9   S nANWS nAff = f! , (       d  f       g = f)Nc                ó
   >• U mg r,   r-   )r„  r{   s    €r=   Ú_get_runÚ_forward.<locals>._get_runµ  s   ø€ à‰r?   c                ó(   >• TR                   U l        g r,   )r\   rw   )r„  r€   s    €r=   Ú_set_reference_example_idÚ+_forward.<locals>._set_reference_example_id¹  s   ø€ Ø!(§¡ˆÕr?   Úexample_version)Úon_endr  r3   r8   rˆ  rw   ÚignoreÚ_on_successz2Unrecognized error_handling value: error_handling=rC  T)rD  Úlangsmith_extraræ  r   zError running target function: r)   )rL  Ú
stacklevel)r{   r€   )r„  z
rt.RunTreerë   rð   )rê  rç  rr  rg   ÚLangSmithExtrar\   r[   r  Ú_get_target_argsrÀ  rã  rU  rR  r]   rG  r©  r   r   rd  )r?  r€   rž   r3   r8   r;   ra  rU   r¬  r¯  r±  rµ  Ú	arg_namesÚargnr‚  rò  ræ  rÍ   r{   s    `                @r=   r<  r<  ©  s„  ù€ ð &*€C÷÷,ð ×*Ñ*×@¨g×.@Ñ.@×KÑKÓM€OÜ×'Ò'ØØ$ØA�HÐAÐ/°ÑAØñ	€Oð ˜ÓØ29·*±*ˆÐ.Ò/Ø	˜8Ó	#à)B˜Ò&äÐN¸~Ñ>OÐPÓQÐQä	×	Ò	¶>¡GÀtÓ	Lð	Ü(¨Ó,ˆIÙ7@ÓA±y¨t”G˜G TÖ*±yˆDÐAÙ�Ð6 oÒ6æ" w×':Ñ':Ñ'FØ")×"5Ô"5�JØ$×0Ñ0°Ñ<¸XÑF�FØ—K‘K –Nñ #6ô ¤4¬¯©°SÓ#9À7ÑK÷ 
MÑ	Lùò Bøô ó 	Ü�L‰LØ1°!°Ð5ÀÐQRð ö ûð	ú÷ 
MÕ	LúsI   Â,FÂ.EÂ=EÃAEÄ&!FÅEÅ
FÅ E=Å8FÅ=FÆFÆ
Fc                óR   •  [         R                  " U 5        g! [         a     gf = frv  rw  rx  s    r=   Ú_is_valid_uuidr¼  Ü  rz  r{  c                óÀ   • U (       d  / $ / n [        U 5       H0  nUR                   H  nU(       d  M  UR                  U5        M     M2     U$ ! [         a    Us $ f = f)aI  Best-effort extraction of feedback keys for the progress hint.

Used to populate ``extra.__progress.evaluator_keys`` on the experiment
session so the UI can render per-evaluator progress. Returns an empty list
when keys cannot be statically inferred (e.g. arbitrary callables whose
return shape isn't visible to AST inspection).
)r(  rS  rÄ   rR  )r1   ÚkeysÚevrF   s       r=   rf  rf  ä  sb   € ö Øˆ	Ø€DðÜ% jÖ1ˆBØ×%Ô%�ß�1Ø—K‘K –Nó &ñ 2ð €Køô ó ØŠðús   �"A ³A ÁAÁAc               ól  •  [        U [        5      (       a   [        U [        5      (       d  [        U 5      $ Sn[        U [        R
                  5      (       a&  U R                  b  U R                  $ U R                  nO][        U [        R                  5      (       a  U nO;[        U [        5      (       a&  [        U 5      (       a  [        R                  " U 5      nUb  UR                  US9R                  $ [        U [        5      (       a  UR                  U S9R                  $ g! [         a     gf = f)uL  Best-effort dataset size for the experiment progress hint.

Returns ``None`` for lazy iterators where size cannot be inferred without
consuming the iterable. Never raises â€” failures (e.g. transient backend
error during ``read_dataset``) degrade to ``None`` and the backend resolves
the count from the dataset at experiment start.
N)r¬   )Údataset_name)rN   r
   rO   rb   r   ÚDatasetÚexample_countr\   rP   rQ   r¼  Úread_datasetrR  )r0   r8   r¬   s      r=   rØ  rØ  û  s÷   € ðä�dœE×"Ñ"¬:°d¼C×+@Ñ+@Ü�t“9Ðà*.ˆ
Ü�dœGŸO™O×,Ñ,Ø×!Ñ!Ñ-Ø×)Ñ)Ð)ØŸ™‰JÜ˜œdŸi™i×(Ñ(Ø‰JÜ˜œc×"Ñ"¤~°d×';Ñ';ÜŸš 4›ˆJØÑ!Ø×&Ñ&°*Ð&Ð=×KÑKÐKä�dœC× Ñ Ø×&Ñ&°DÐ&Ð9×GÑGÐGØøÜó Ùðús#   ‚4D& ·9D& Á1BD& Ã7-D& Ä&
D3Ä2D3)ra  c               ó®  • [        U [        R                  5      (       a  UR                  XS9$ [        U [        5      (       a4  [        U 5      (       a$  UR                  [        R                  " U 5      US9$ [        U [        5      (       a  UR                  XS9$ [        U [        R                  5      (       a  UR                  U R                  US9$ U $ )z*Return the examples for the given dataset.)r¬   ra  )rÁ  ra  )	rN   rP   rQ   r*  rO   r¼  r   rÂ  r\   )r0   r8   ra  s      r=   rù  rù    sÕ   € ô �$œŸ	™	×"Ñ"Ø×#Ñ#Øð $ð 
ð 	
ô 
�Dœ#×	Ñ	¤>°$×#7Ñ#7Ø×#Ñ#Ü—y’y “Ð<Oð $ð 
ð 	
ô 
�Dœ#×	Ñ	Ø×#Ñ#Øð $ð 
ð 	
ô 
�Dœ'Ÿ/™/×	*Ñ	*Ø×#Ñ#Ø—w‘wÐ4Gð $ð 
ð 	
ð €Kr?   c                ó   • SU ;   a  U S   $ U $ )Nrè  r-   )rè  s    r=   Ú_default_process_inputsrÇ  9  s   € Ø'¨6Ó1ˆ6�(ÑÐ=°vÐ=r?   c                ó  • [        U 5      (       d  [        S5      e[        R                  " U 5      (       a  U nU$ [	        U 5      (       a  U R
                  n [        R                  " S[        S9" [        [        U 5      5      nU$ )z(Ensure the target function is traceable.zÐTarget must be a callable function or a langchain/langgraph object. For example:

def predict(inputs: dict) -> dict:
    # do work, like chain.invoke(inputs)
    return {...}

evaluate(
    predict,
    ...
)ÚTarget)r#  Úprocess_inputs)
r^  r[   rg   Úis_traceable_functionr]  Úinvoker   rÇ  r   r   )r/   r?  s     r=   r;  r;  =  s}   € ô ˜×ÑÜðó

ð 
	
ô 
×Ò ×'Ñ'Ø6<ˆð €Iô " &×)Ñ)Ø—]‘]ˆFÜ�\Š\˜xÔ8OÒPÜ”˜6Ó"ó
ˆð €Ir?   c                óN   • [        U 5      =(       d    [        [        U5      5      $ r,   )r  rV   Ú_evaluators_include_attachments)r/   r1   s     r=   rg  rg  Y  s#   € Ü& vÓ.÷ ´$Ü'¨
Ó3ó3ð r?   c                ó.   • U c  g[        S U  5       5      $ )Nr   c              3  ó6   #   • U H  n[        U5      v •  M     g 7fr,   )Ú_evaluator_uses_attachments)rE   rÍ   s     r=   rH   Ú2_evaluators_include_attachments.<locals>.<genexpr>c  s   é € ÐB±z°!Ô*¨1×-Ð-²zùs   ‚)Úsum)r1   s    r=   rÎ  rÎ  _  s   € ØÑØäÑB±zÓBÓBÐBr?   c                ó4  • [        U 5      (       d  g[        R                  " U 5      n[        UR                  R                  5       5      nU Vs/ sH,  o3R                  UR                  UR                  4;   d  M*  UPM.     nn[        S U 5       5      $ s  snf )NFc              3  ó<   #   • U H  oR                   S :H  v •  M     g7f)rã  Nr˜  )rE   r1  s     r=   rH   Ú._evaluator_uses_attachments.<locals>.<genexpr>n  s   é € ÐBÑ0A¨1�v‰v˜Ö&Ò0Aùs   ‚)
rf   ÚinspectÚ	signaturera   Ú
parametersrX   ÚkindÚPOSITIONAL_ONLYÚPOSITIONAL_OR_KEYWORDrW   )r>  ÚsigÚparamsr1  Úpositional_paramss        r=   rÑ  rÑ  f  s†   € Ü�I×ÑØÜ
×
Ò
˜IÓ
&€CÜ�#—.‘.×'Ñ'Ó)Ó*€FáóÙˆaŸV™V¨×(9Ñ(9¸1×;RÑ;RÐ'SÑS�‘6ð ð ô ÑBÑ0AÓBÓBÐBùòs   Á(BÁ;Bc                ó   • S[        U 5      ;   $ )ú0Whether the target function accepts attachments.rã  )r¸  r‹   s    r=   r  r  q  s   € àÔ,¨VÓ4Ñ4Ð4r?   c                óp  • [        U 5      (       d  / $ [        U 5      (       a  S/$ [        R                  " U 5      n[	        UR
                  R                  5       5      nU Vs/ sH,  o3R                  UR                  UR                  4;   d  M*  UPM.     nnU Vs/ sH  o3R                  UR                  L d  M  UPM!     nn[        U5      S:X  a  [        S5      e[        U5      S:”  a  [        S5      e[        U5      S:”  aX  U Vs1 sH  o3R                  iM     snR                  / SQ5      (       a'  [        SU Vs/ sH  o3R                  PM     sn 35      e/ nUS	S  H0  nUR                  S
;   a  UR!                  UR                  5        M0    O   U=(       d    S/$ s  snf s  snf s  snf s  snf )rá  rè  r   zFTarget function must accept at least one positional argument (inputs).r"  zlTarget function must accept at most three arguments without default values: (inputs, attachments, metadata).r)   )rè  rã  r3   zˆWhen passing multiple positional arguments without default values, they must be named 'inputs', 'attachments', or 'metadata'. Received: N>   rè  r3   rã  )rf   r]  r×  rØ  ra   rÙ  rX   rÚ  rÛ  rÜ  Údefaultrº   rb   r[   r#  Ú
differencerÄ   )r/   rÝ  rÞ  r1  rß  Úpositional_no_defaultr‚  s          r=   r¸  r¸  v  s²  € ä�F×ÑØˆ	Ü˜f×%Ñ%ØˆzÐä
×
Ò
˜FÓ
#€CÜ�#—.‘.×'Ñ'Ó)Ó*€FáóÙˆaŸV™V¨×(9Ñ(9¸1×;RÑ;RÐ'SÑS�‘6ð ð ñ ):ÓRÑ(9 1¿Y¹YÈ!Ï'É'Ð=QŸQÑ(9ÐÐRä
ÐÓ Ó"ÜØTó
ð 	
ô 
Ð"Ó	# aÓ	'ÜðQó
ð 	
ô 
Ð"Ó	# aÓ	'Ù-ó-Ù-�1�ŒÑ-ñ-ç�jÒ6Ó7õ-8ô ðOá 5Ó6Ñ 5˜1—”Ñ 5Ñ6Ð7ð9ó
ð 	
ð ˆØ" 2 AÓ&ˆAØ�v‰vÐ>Ó>Ø—‘˜AŸF™FÖ#áñ	 'ð
 ×!˜�zÐ!ùò;ùò Sùò-ùò 7s$   Á#(F$ÂF$ÂF)Â:F)Ä	F.Å F3
c                ó–  • U bh  [        U [        R                  5      (       a  U nO[        X5      nUR                  (       d  [        S5      eUR                  (       d  [        S5      eX14$ Ub[  [        R                  " U5      u  pA[        U5      nUR                  UR                  S9nUR                  (       d  [        S5      eX14$ g)Nz,Experiment name must be defined if provided.zOExperiment must have an associated reference_dataset_id, but none was provided.r}  z,Experiment name not found for provided runs.)NN)rN   r   rR   rt   r#  r[   r"  rú  rû  r  r  r	  )r:   ry   r8   rq  rš  Ú	first_runs         r=   re  re  Ÿ  sÀ   € ð ÑÜ�j¤'×"7Ñ"7×8Ñ8Ø$‰Kä*¨:Ó>ˆKà××ÜÐKÓLÐLØ×/×/Üð)óð ð Ð Ð àÑÜ—m’m DÓ)‰ˆÜ˜“Kˆ	Ø×)Ñ)°Y×5IÑ5IÐ)ÐJˆØ××ÜÐKÓLÐLØÐ Ð Ør?   c                 ó   • SSK Jn   U " 5       $ )Nr   ©Úrandom_name)Ú%langsmith.evaluation._name_generationrê  ré  s    r=   r¤  r¤  À  s   € ÝAá‹=Ðr?   c                ó|   •  SS K nUR                  " [        XUS95      $ ! [         a  n[        S5      UeS nAff = f)Nr   z¤The 'pandas' library is required to use the 'to_pandas' function. Please install it using 'pip install pandas' or 'conda install pandas' before calling this method.rÓ   )rÙ   rœ  rô   Ú_flatten_experiment_results)rÊ   r˜   rÔ   ÚpdrÍ   s        r=   rÕ   rÕ   Æ  sM   € ð
Ûð �<Š<Ô3°GÈcÑRÓSÐSøô ó ÜðAó
ð ð		ûðús   ‚   
;ª6¶;c                ót  • XU  VVVVs/ sGH„  n0 US   R                   =(       d    0 R                  5        VVs0 sH  u  pESU 3U_M     snnEUS   R                  =(       d    0 R                  5        VVs0 sH  u  pESU 3U_M     snnESUS   R                  0EUS   R                  b5  US   R                  R                  5        VVs0 sH  u  pESU 3U_M     snnO0 EUS   S    Vs0 sH6  nS	UR                   3UR
                  b  UR
                  OUR                  _M8     snEUS   R                  (       a-  US   R                  US   R                  -
  R                  5       OS US   R                  US   R                  S
.EPGM‡     snnnn$ s  snnf s  snnf s  snnf s  snf s  snnnnf )Nr€   zinputs.r{   zoutputs.rG  z
reference.r�   rÊ   r   )Úexecution_timerA  r\   )rè  rZ   ré  rG  r  r  rS  Úend_timer
  Útotal_secondsrw   r\   )rÊ   r˜   rÔ   r™  rF   rG   r„  s          r=   rí  rí  ×  sÊ  € ð6 ˜sÑ#ö-ò, $ˆAð+	
Ø-.¨y©\×-@Ñ-@×-FÀB×,MÑ,MÔ,OÔPÑ,O¡D A�˜˜ˆ}˜aÒÑ,OÒPð	
à./°©h×.>Ñ.>×.DÀ"×-KÑ-KÔ-MÔNÑ-M¡T Q�˜!˜ˆ~˜qÒ Ñ-MÒNð	
ð �Q�u‘X—^‘^ñ	
ð �Y‘<×'Ñ'Ñ3ð 23°9±×1EÑ1E×1KÑ1KÔ1MÔNÑ1M©¨�:˜a˜SÐ! 1Ò$Ñ1MÓNàð	
ð Ð/Ñ0°Ò;óá;�Að ˜AŸE™E˜7Ð#°·±Ñ0C Q§W¢WÈÏÉÒPÙ;ñð	
ð �U‘8×$×$ð �5‘×"Ñ" Q u¡X×%8Ñ%8Ñ8×GÑGÔIàà˜E™(×7Ñ7Ø�E‘(—+‘+ö'	
ñ* $ô-ð ùãPùÛNùó Oùòùõs<   Š0F2
ºFÁ-F2
Á8F!Â	AF2
ÃF'ÃF2
Ã/<F-Ä+A*F2
ÆF2
)Úmaxsizec                 ó4   •  SSK Jn   U $ ! [         a     g f = f)Nr   r'   )Úlangchain_core.runnablesr(   rœ  r'   s    r=   Ú_import_langchain_runnablerö  ö  s!   € ðÝ5àˆøÜó Ùðús   ‚
 Š
–c                óP   • [        [        5       =n=(       a    [        X5      5      $ r,   )rV   rö  rN   )Úor(   s     r=   r]  r]   	  s    € ÜÔ7Ó9Ð9�×V¼zÈ!Ó?VÓWÐWr?   )NNNNNNr   r)   NTNT)r0   úOptional[DATA_T]r1   úOptional[Sequence[EVALUATOR_T]]r2   ú'Optional[Sequence[SUMMARY_EVALUATOR_T]]r3   rÏ  r4   rî   r5   rî   r6   ró   r7   rñ   r8   rÐ  r9   rV   r:   úOptional[EXPERIMENT_T]r;   rV   r/   z'Union[TARGET_T, Runnable, EXPERIMENT_T]r<   r   rë   r*   )r0   rù  r1   z+Optional[Sequence[COMPARATIVE_EVALUATOR_T]]r2   rû  r3   rÏ  r4   rî   r5   rî   r6   ró   r7   rñ   r8   rÐ  r9   rV   r:   rü  r;   rV   r/   z(Union[tuple[EXPERIMENT_T, EXPERIMENT_T]]r<   r   rë   r@   )NNNNNNr   r)   NTNTrˆ  ) r0   rù  r1   zIOptional[Union[Sequence[EVALUATOR_T], Sequence[COMPARATIVE_EVALUATOR_T]]]r2   rû  r3   rÏ  r4   rî   r5   rî   r6   ró   r7   rñ   r8   rÐ  r9   rV   r:   rü  r;   rV   rU   rŒ  r/   zJUnion[TARGET_T, Runnable, EXPERIMENT_T, tuple[EXPERIMENT_T, EXPERIMENT_T]]r<   r   rë   z6Union[ExperimentResults, ComparativeExperimentResults])NNNr   NFT)r1   rú  r2   rû  r3   rÏ  r6   ró   r8   rÐ  rq   rV   r9   rV   r:   ú,Union[str, uuid.UUID, schemas.TracerSession]rë   r*   )NNé   NNFF)r1   z!Sequence[COMPARATIVE_EVALUATOR_T]r4   rî   r5   rî   r6   rñ   r8   rÐ  r3   rÏ  rq   rV   r  rV   rû   z!tuple[EXPERIMENT_T, EXPERIMENT_T]rë   r@   )rû   z3tuple[schemas.TracerSession, schemas.TracerSession]r  zschemas.ComparativeExperimentrë   rî   )r´   rî   rë   rð   )r/   ú0Union[TARGET_T, Iterable[schemas.Run], Runnable]rë   rV   )NNNNNNr)   NTNTrˆ  )r0   r�  r1   rú  r2   rû  r3   rÏ  r4   rî   r5   rî   r6   ró   r7   rñ   r8   rÐ  r9   rV   r:   ú6Optional[Union[schemas.TracerSession, str, uuid.UUID]]r;   rV   rU   rŒ  r/   rÿ  rë   r*   )rS  rO   rë   rV   )rx   ÚEXPERIMENT_Tr8   úlangsmith.Clientrë   rÑ  )F)rx   rý  r8   r  rq   rV   rë   r!  )r8   r  rx   rÑ  rë   z dict[uuid.UUID, schemas.Example])rë   zCallable[[IT], IT])r1   z8Sequence[Union[EVALUATOR_T, RunEvaluator, AEVALUATOR_T]]rë   r‘  )r1   r�  rë   zlist[SUMMARY_EVALUATOR_T])Frˆ  )r?  zrh.SupportsLangsmithExtrar€   r   rž   rO   r3   rV  r8   r  r;   rV   ra  rV   rU   rŒ  rë   r©  )r1   zBOptional[Sequence[Union[EVALUATOR_T, RunEvaluator, AEVALUATOR_T]]]rë   ú	list[str])r0   z-Union[DATA_T, AsyncIterable[schemas.Example]]r8   r  rë   ró   )r0   r�  r8   r  ra  rV   rë   rŽ  )rè  rV  rë   rV  )r/   z=TARGET_T | rh.SupportsLangsmithExtra[[dict], dict] | Runnablerë   z'rh.SupportsLangsmithExtra[[dict], dict])r/   r   r1   úOptional[Sequence]rë   rV   )r1   r  rë   rñ   )r>  r   rë   rV   )r/   r   rë   rV   )r/   r   rë   r  )r:   r   ry   r‰  r8   r  rë   zStuple[Optional[Union[schemas.TracerSession, str]], Optional[Iterable[schemas.Run]]]rê   rò   )rÊ   zlist[ExperimentResultRow]r˜   ró   rÔ   ró   )rë   zOptional[type])rø  r   rë   rV   )~rõ   Ú
__future__r   r+  Úconcurrent.futuresr@  r/  r¤  r×  rî  rú  Úloggingri  rŽ   r  r‘   rP   Úcollections.abcr   r   r   r   r   r	   r
   Úcontextvarsr   Útypingr   r   r   r   r   r   r   r   Útyping_extensionsr   r   Ú	langsmithr   r¦  r   rg   r   rr   r   r   r-  Úlangsmith._internalr   Ú#langsmith._internal._beta_decoratorr   r   Úlangsmith.evaluation.evaluatorr   r   r    r!   r"   r#   r$   r%   r&   rÙ   rî  rõ  r(   rô   Ú	getLoggerrƒ   r]   rV  rn  rO   rQ   rð  rÂ  r�  rd  r`   ÚAEVALUATOR_TrR   r  r>   r_   r}   r*   re   rd   r@   r'  r(  r^  ri   ry  rt   ru   rv   r•  rÁ   r�  rž  ré   r(  r.  r©  r<  r¼  rf  rØ  rù  rÇ  r;  rg  rÎ  rÑ  r  r¸  re  r¤  rÕ   rí  Ú	lru_cacherö  r]  r-   r?   r=   Ú<module>r     sT	  ðÙ å "ã Ý Û Û Û 	Û Û Û Û Û Û Û ÷÷ ñ õ %÷	÷ 	ó 	÷ 2ã Ý #Ý 'Ý %Ý Ý 'Ý 3÷÷
÷ 
õ 
ö ÛÝ1à—‘�Ià€IØ	×	Ò	˜8Ó	$€à�˜4˜& $˜,Ñ'¨°4¸°,ÀÐ2DÑ)EÐEÑF€à	ˆs�D—I‘I˜x¨¯©Ñ8¸'¿/¹/ÐIÑ	J€ð ØØØ	�‰�h˜wŸ™Ñ/Ð0ØÐÐ 1Ð1Ñ2ð	4ñð ˆS�%˜Ð/Ð1AÐAÑBÐBÑCðEñ€ð ØØ	�‰�h˜wŸ™Ñ/Ð0Ø�%Ð(Ð*;Ð;Ñ<Ñ=ð	?ñð ñ€ð �S˜$Ÿ)™) W×%:Ñ%:Ð:Ñ;€ð 
ð "Ø26ØBFØ#Ø'+Ø!%Ø%&ØØ)-ØØ)-Øðð ðð 0ð	ð
 @ðð ðð %ðð ðð #ðð ðð 'ðð ðð 'ðð ðØ3ðð ðð  ô!ó 
ðð& 
ð "Ø>BØBFØ#Ø'+Ø!%Ø%&ØØ)-ØØ)-Øð'ð ð'ð <ð	'ð
 @ð'ð ð'ð %ð'ð ð'ð #ð'ð ð'ð 'ð'ð ð'ð 'ð'ð ð'Ø4ð'ð ð'ð  "ô!'ó 
ð'ð, "ð 	ØBFØ#Ø'+Ø!%Ø%&ØØ)-ØØ)-ØØ/4ð#V
ð ðV
ðð	V
ð @ðV
ð ðV
ð %ðV
ð ðV
ð #ðV
ð ðV
ð 'ðV
ð ðV
ð 'ðV
ð  ð!V
ð" -ð#V
ØVðV
ð$ ð%V
ð& <õ'V
ðx 37ØBFØ#Ø%&Ø)-ØØðmð 0ðmð @ð	mð
 ðmð #ðmð 'ðmð ðmð ðmØ<ðmð õmô`*˜)ô *÷z)ñ z)ð@ #Øˆg�k‰kÑ˜H W§_¡_Ñ5Ð6Ø	ØÐ(¨$Ð.Ñ/Ø�%Ð2°DÐ8Ñ9Ñ:ð	<ñðñÐ ð (,Ø!%ØØ)-Ø#ØØ!ðNð 2ðNð %ð	Nð
 ðNð ðNð 'ðNð ðNð ðNð ðNØ2ðNð "õN÷b
,ñ ,ðdØDðà9ðð ôð"
Ø!ð
à	ô
ô>ð 37ØBFØ#Ø'+Ø!%Ø%)ØØ)-ØØIMØØ/4ð9ð ð9ð 0ð	9ð
 @ð9ð ð9ð %ð9ð ð9ð #ð9ð ð9ð 'ð9ð ð9ð Gð9ð ð9ð -ð9Ø<ð9ð  õ!9ôxð9Øð9Ø#3ð9àô9ð ð*Ø9ð*àð*ð ð*ð õ	*ðZ	Øð	Ø'<ð	à%ô	ñ ˆTƒ]€ôñ ˆTÐ2Ñ3€÷sñ sôll9Ð0ô l9ð^	ØHð	àô	ðØ-ðàôô:�iô ð !&Ø/4ð0LØ!ð0Làð0Lð ð0Lð ð	0Lð
 ð0Lð ð0Lð ð0Lð -ð0Lð õ0LôfðØRðàôð.!Ø
7ð!ð ð!ð ô	!ðP !&ñ	Ø
ðð ðð ð	ð
 õô4>ðØIðà,ôô8ôCôCô5ô
&"ðRØFðà
)ðð ððô	ôBð ØðTØ&ðTàðTð 
õTð& ØðØ&ðàðð 
õð> ×Ò˜QÑóó  ðõXr?   