ó
    üÞ jH#  ã                   ó(  • S r SSKrSSKrSSKrSSKrSSKJr  SSKJrJ	r	  SSK
Jr  SSKJr  SSKJr  SSKJr  SSKJrJrJr  SSKJr  S	\S
\S\4S jrS\R:                  S\S\\   4S jr\SSSSS.S\\R:                     S\ S\\    S\\   S\!S\!S\RD                  4S jj5       r#S\ S\S\\R:                     4S jr$\	" S5      r%\	" S5      r&S\\%   S\\&   S\\'\%\&4      4S jr(\" S 5      \S!SS".S\ S#\S$\\)   S\\   SS4
S% jj5       5       r*g)&zfBeta utility functions to assist in common eval workflows.

These functions may change in the future.
é    N)ÚSequence)ÚOptionalÚTypeVar)Ú
evaluation)Ú_v2_migration_utils)Ú
deprecatedÚsuppress_deprecation_warningÚ	warn_beta)ÚClientÚrun_dictÚid_mapÚreturnc                 ó  • U S   nUR                  5        H)  u  p4UR                  [        U5      [        U5      5      nM+     X S'   U R                  S5      (       a
  XS      U S'   U R                  S5      (       d  0 U S'   U $ )zàConvert the IDs in the run dictionary using the provided ID map.

Parameters:
- run_dict: The dictionary representing a run.
- id_map: The dictionary mapping old IDs to new IDs.

Returns:
- dict: The updated run dictionary.
Údotted_orderÚparent_run_idÚextra)ÚitemsÚreplaceÚstrÚget)r   r   ÚdoÚkÚvs        ÚO/var/www/html/gaurav/venv/lib/python3.13/site-packages/langsmith/beta/_evals.pyÚ_convert_idsr      s   € ð 
�.Ñ	!€BØ—‘–‰ˆØ�Z‰Zœ˜A›¤ A£Ó'Šñ à!ˆ^Ñà‡|�|�O×$Ñ$Ø$*°OÑ+DÑ$Eˆ�Ñ!Ø�<‰<˜× Ñ Øˆ�ÑØ€Oó    ÚrootÚrun_to_example_mapc                 ó  • U /n[         R                  " 5       nU R                  U0n/ nU(       a¨  UR                  5       nUR	                  1 SkS9nUR                  US   [         R                  " 5       5      XGS   '   XGS      US'   XGS      US'   UR                  (       a  UR                  UR                  5        UR                  U5        U(       a  M¨  U Vs/ sH  n[        X„5      PM     n	nXR                     U	S   S'   U	$ s  snf )zêConvert the root run and its child runs to a list of dictionaries.

Parameters:
- root: The root run to convert.
- run_to_example_map: The dictionary mapping run IDs to example IDs.

Returns:
- The list of converted run dictionaries.
>   Ú
session_idÚchild_run_idsÚparent_run_ids)ÚexcludeÚidÚtrace_idr   Úreference_example_id)ÚuuidÚuuid4r%   ÚpopÚdictr   Ú
child_runsÚextendÚappendr   r$   )
r   r   Úruns_r%   r   ÚresultsÚsrcÚsrc_dictÚrÚresults
             r   Ú_convert_root_runr4   /   sð   € ð ˆF€EÜ�zŠz‹|€HØ�m‰m˜XÐ&€FØ€GÞ
Ø�i‰i‹kˆØ—8‘8Ò$U�8ÐVˆØ!'§¡¨H°T©N¼D¿JºJ»LÓ!Iˆ˜‰~ÑØ¨¡Ñ/ˆ�‰Ø%¨zÑ&:Ñ;ˆ�ÑØ�>�>Ø�L‰L˜Ÿ™Ô(Ø�‰�xÔ ÷ ˆ%ñ 07Ó7©w¨!Œl˜1Ö%©w€FÐ7Ø(:¿7¹7Ñ(C€Fˆ1�IÐ$Ñ%Ø€Mùò 8s   ÃDF)Útest_project_nameÚclientÚload_child_runsÚinclude_outputsÚrunsÚdataset_namer5   r6   r7   r8   c                óê  • U (       d  [        SU  35      eU=(       d    [        R                  " 5       nUR                  US9nU(       a  U  Vs/ sH  owR                  PM     snOSnUR                  U  Vs/ sH  owR                  PM     snUU  Vs/ sH  owR                  PM     snUR                  S9  U(       d  U n	OU  Vs/ sH  osR                  U5      PM     n	nU=(       d%    S[        R                  " 5       R                  SS  3n[        UR                  US95      n
U
 Vs0 sH  o»R                  UR                  _M     nnU
S   R                  (       a  U
S   R                  OU
S   R                   nU	 VVs/ sH  n[#        Xì5       H  nUPM     M     nnnUR%                  UUR                  SUR'                  5       S	.S
9nU Hg  nUS   US   -
  n[(        R(                  R+                  [(        R,                  R.                  S9US'   US   U-   US'   UR0                  " S0 UDSU0D6  Mi     UR3                  UR                  5      nU$ s  snf s  snf s  snf s  snf s  snf s  snnf )aP  Convert the following runs to a dataset + test.

This makes it easy to sample prod runs into a new regression testing
workflow and compare against a candidate system.

Internally, this function does the following:
    1. Create a dataset from the provided production run inputs.
    2. Create a new test project.
    3. Clone the production runs and re-upload against the dataset.

Parameters:
- runs: A sequence of runs to be executed as a test.
- dataset_name: The name of the dataset to associate with the test runs.
- client: An optional LangSmith client instance. If not provided, a new client will
    be created.
- load_child_runs: Whether to load child runs when copying runs.

Returns:
- The project containing the cloned runs.

Example:
--------
```python
import langsmith
import random

client = langsmith.Client()

# Randomly sample 100 runs from a prod project
runs = list(client.list_runs(project_name="My Project", execution_order=1))
sampled_runs = random.sample(runs, min(len(runs), 100))

runs_as_test(runs, dataset_name="Random Runs")

# Select runs named "extractor" whose root traces received good feedback
runs = client.list_runs(
    project_name="<your_project>",
    filter='eq(name, "extractor")',
    trace_filter='and(eq(feedback_key, "user_score"), eq(feedback_score, 1))',
)
runs_as_test(runs, dataset_name="Extraction Good")
```
z1Expected a non-empty sequence of runs. Received: )r:   N)ÚinputsÚoutputsÚsource_run_idsÚ
dataset_idzprod-baseline-é   r   zprod-baseline)ÚwhichÚdataset_version)Úproject_nameÚreference_dataset_idÚmetadataÚend_timeÚ
start_time)ÚtzrC   © )Ú
ValueErrorÚrtÚget_cached_clientÚcreate_datasetr=   Úcreate_examplesr<   r$   Ú_load_child_runsr'   r(   ÚhexÚlistÚlist_examplesÚsource_run_idÚmodified_atÚ
created_atr4   Úcreate_projectÚ	isoformatÚdatetimeÚnowÚtimezoneÚutcÚ
create_runÚupdate_project)r9   r:   r5   r6   r7   r8   Údsr2   r=   Úruns_to_copyÚexamplesÚer   rB   Úroot_runr   Ú	to_createÚprojectÚnew_runÚlatencyÚ_s                        r   Úconvert_runs_to_testrh   K   sk  € öj ÜÐNÈtÈfÐWÓXÐXØ×-”r×+Ò+Ó-€FØ	×	Ñ	¨LÐ	Ð	9€BÞ+:¡$Ó'¡$˜Q�yŒy¡$Ò'À€GØ
×ÑÙ"&Ó'¡$˜Q—”¡$Ñ'ØÙ&*Ó+¡d Ÿœ¡dÑ+Ø—5‘5ð	 ñ ö Ø‰á<@ÓA¹D°q×/Ñ/°Ö2¹DˆÐAà)×T¨~¼d¿jºj»l×>NÑ>NÈrÐPQÐ>RÐ=SÐ-TÐä�F×(Ñ(°lÐ(ÐCÓD€HÙ9AÓB¹°AŸ/™/¨1¯4©4Ò/¹ÐÐBà#+¨A¡;×#:×#:ˆ�‰×ÒÀÈÁ×@VÑ@Vð ñ %ôá$ˆHÜ)¨(ÖGˆHó 	áGñ 	Ù$ð ñ ð ×#Ñ#Ø&ØŸU™Uà$Ø.×8Ñ8Ó:ñ
ð $ð €Gó ˆØ˜*Ñ%¨°Ñ(=Ñ=ˆÜ (× 1Ñ 1× 5Ñ 5¼×9JÑ9J×9NÑ9NÐ 5Ð Oˆ�ÑØ% lÑ3°gÑ=ˆ�
ÑØ×ÒÑD˜GÑDÐ2CÕDñ	 ð 	×ÑØ�
‰
ó	€Að €Nùò[ (ùâ'ùâ+ùò Bùò
 Cùó
s$   ÁIÁ6IÂI ÃI%Ä(I*Æ I/rC   c                 ó`  • [         R                  " UR                  R                  5      nU[         R                  R
                  :X  a  [         R                  " X5      $ [        5          UR                  U S9nS S S 5        [        R                  " [        5      n/ n0 nW HM  nUR                  b  XGR                     R                  U5        OUR                  U5        XvUR                  '   MO     UR                  5        H  u  p‰[!        U	S S9Xh   l        M     U$ ! , (       d  f       N¬= f)N)rC   c                 ó   • U R                   $ ©N)r   )r2   s    r   Ú<lambda>Ú%_load_nested_traces.<locals>.<lambda>Ë   s   € ÀqÇ~Â~r   )Úkey)r   Úget_query_backendÚinfoÚinstance_flagsÚQueryBackendÚSMITHDB_ONLYÚ_load_nested_traces_v2r	   Ú	list_runsÚcollectionsÚdefaultdictrQ   r   r-   r$   r   Úsortedr+   )
rC   r6   Úbackendr9   Útreemapr/   Úall_runsÚrunÚrun_idr+   s
             r   Ú_load_nested_tracesr~   ´   s   € ô "×3Ò3°F·K±K×4NÑ4NÓO€GØÔ%×2Ñ2×?Ñ?Ó?Ü"×9Ò9¸,ÓOÐOô 
&Õ	'Ø×Ñ¨\ÐÐ:ˆ÷ 
(ô 	×Ò¤Ó%ð ð €GØ€HÛˆØ×ÑÑ(Ø×%Ñ%Ñ&×-Ñ-¨cÕ2à�N‰N˜3ÔØ�—‘Óñ ð &Ÿm™mžoÑˆÜ&,¨ZÑ=UÑ&VˆÑÖ#ñ .à€N÷ 
(Õ	'ús   Á)DÄ
D-ÚTÚUÚlist1Úlist2c                 ó@   • [        [        R                  " X5      5      $ rk   )rQ   Ú	itertoolsÚproduct)r�   r‚   s     r   Ú_outer_productr†   Ó   s   € Ü”	×!Ò! %Ó/Ó0Ð0r   z´compute_test_metrics() is deprecated and will be removed after Jan 31, 2027. There is no replacement: run the evaluators yourself and log the results with client.create_feedback().é
   )Úmax_concurrencyr6   Ú
evaluatorsrˆ   c          
      ó  • SSK Jn  / nU H�  n[        U[        R                  5      (       a  UR                  U5        M5  [        U5      (       a'  UR                  [        R                  " U5      5        Ml  [        S[        U5       35      e   U=(       d    [        R                  " 5       n[        X5      nU" US9 nUR                  " UR                  /[        [!        Xu5      6 Q76 n	SSS5        W	 H  n
M     g! , (       d  f       N= f)a²  Compute test metrics for a given test name using a list of evaluators.

.. admonition:: Deprecated

    There is no replacement: run the evaluators yourself and log the results
    with :meth:`langsmith.Client.create_feedback`.
    Will be removed after Jan 31, 2027.

Args:
    project_name (str): The name of the test project to evaluate.
    evaluators (list): A list of evaluators to compute metrics with.
    max_concurrency (Optional[int], optional): The maximum number of concurrent
        evaluations. Defaults to 10.
    client (Optional[Client], optional): The client to use for evaluations.
        Defaults to None.

Returns:
    None: This function does not return any value.
r   )ÚContextThreadPoolExecutorz5Evaluation not yet implemented for evaluator of type )Úmax_workersN)Ú	langsmithr‹   Ú
isinstanceÚls_evalÚRunEvaluatorr-   ÚcallableÚrun_evaluatorÚNotImplementedErrorÚtyperK   rL   r~   ÚmapÚevaluate_runÚzipr†   )rC   r‰   rˆ   r6   r‹   Úevaluators_ÚfuncÚtracesÚexecutorr/   rg   s              r   Úcompute_test_metricsrœ   ×   sæ   € õ@ 4à.0€KÛˆÜ�dœG×0Ñ0×1Ñ1Ø×Ñ˜tÖ$Ü�d�^‰^Ø×Ñœw×4Ò4°TÓ:Ö;ä%ØGÌÈTË
À|ÐTóð ñ ð ×-”r×+Ò+Ó-€FÜ  Ó6€FÙ	"¨Ò	?À8Ø—,’,Ø×Ñð
Ü"%¤~°fÓ'JÐ"Kò
ˆ÷ 
@ó ˆÙò ÷	 
@Õ	?ús   Â?-C>Ã>
D)+Ú__doc__rv   rX   r„   r'   Úcollections.abcr   Útypingr   r   Úlangsmith.run_treesÚ	run_treesrK   Úlangsmith.schemasÚschemasÚ
ls_schemasr�   r   r�   Úlangsmith._internalr   Ú#langsmith._internal._beta_decoratorr   r	   r
   Úlangsmith.clientr   r*   r   ÚRunrQ   r4   r   ÚboolÚTracerSessionrh   r~   r   r€   Útupler†   Úintrœ   rI   r   r   Ú<module>r­      sÌ  ðñó
 Û Û Û Ý $ß $å  Ý &Ý +Ý 3÷ñ õ
 $ð˜4ð ¨ð °$ô ð,˜JŸN™Nð Àð ÈÈdÉô ð8 ð
 (,Ø#Ø!Ø!òeØ
�:—>‘>Ñ
"ðeð ðeð   ‘}ð	eð
 �VÑðeð ðeð ðeð ×Ñôeó ðeðP cð °6ð ¸dÀ:Ç>Á>Ñ>Rô ñ6 ˆCƒL€ÙˆCƒL€ð1˜$˜q™'ð 1¨$¨q©'ð 1°d¸5ÀÀAÀ¹;Ñ6Gô 1ñ ð%óð
 ð
 &(Ø#ò-Øð-ð ð-ð ˜c‘]ð	-ð
 �VÑð-ð 
ô-ó óñ-r   